{ "benchmark": "JevBench", "revision": "v1.3.0", "revision_log": [ { "revision": "v1.3.0", "date": "2026-09-22", "note": "Scoring-only release; the task set and measurements are unchanged. Intelligence is now accuracy above each item's uniform-guessing baseline, aggregated with the existing tier weights. If chance-corrected Intelligence is below 50, the composite receives a growing (Intelligence / 50)^2 penalty. Calibration, Speed, Cost and ranking eligibility are unchanged." }, { "revision": "v1.2.16", "date": "2026-09-21", "note": "Added the reranker class and five Apache-2.0 open rerankers. A neutral adapter, identical task instructions, public-only temperature/yes-no calibration grids, and a no-instruction public baseline were preregistered before the held-out run. Every row covers all 534 frozen decisions; no earlier row or task changed." }, { "revision": "v1.2.15", "date": "2026-09-21", "note": "Added Open-Jev 2B and Open-Jev 9B by Zefan Cai. Both public checkpoints ran all 534 frozen decisions through the author's pinned LoRA-plus-decision-head server, serially on our RunPod H100 with prefix caching off. An exact normalized-text audit found no JevBench public state or instruction in the 79,116-row public training projection. No earlier measurement or task changed." }, { "revision": "v1.2.14", "date": "2026-09-21", "note": "Added Winnow-12B Q8 after its author requested evaluation. The pinned Q8_0 GGUF ran all 534 unchanged decisions through the author's pinned TypeSafe-compatible server, serially on our lium.io RTX 4090. The row records the applicable Apache-2.0 Gemma 4 terms, non-zero hosted-reference cost and the independently unverifiable private-training overlap claim. No earlier measurement or task changed." }, { "revision": "v1.2.13", "date": "2026-09-21", "note": "Added OpenJev (thinking, BF16) using OpenJev's native typed-API think=512 switch over DiffusionGemma. It ran all 534 unchanged decisions serially on the same H200. No earlier measurement or task changed." }, { "revision": "v1.2.12", "date": "2026-09-21", "note": "Added djev (thinking), an experimental full-generation run over the same open DiffusionGemma checkpoint used by djev-dev. Thinking was enabled with a frozen 8,192-token output cap on all 534 unchanged decisions. Current djev-dev itself hard-codes thinking off, one denoising step and read-only inference, so this row is not presented as a switch in its published typed API. No earlier measurement or task changed." }, { "revision": "v1.2.10", "date": "2026-09-21", "note": "Added smalljev semantic-v9 after its author requested evaluation. The public Apache-2.0 MiniCPM5-2B-Base LoRA and native decision heads ran all 534 frozen decisions serially on our lium.io A6000. Its mapping, endpoint condition, non-zero hosted-reference cost basis and public-benchmark-directed training disclosure were committed before the run (docs/v1.2-additions-smalljev.md). No earlier measurement or task changed." }, { "revision": "v1.2.9", "date": "2026-09-21", "note": "Added Certo v1 (AltSlate Labs) after its author requested evaluation. The public MIT ModernBERT-large checkpoint ran all 534 frozen decisions through the author's DecisionModel, serially on our RunPod RTX 3090. Its mapping, published 64-token state and 48-token option limits, endpoint condition and hosted-price cost basis were pushed before the run (docs/v1.2-additions-certo.md). No earlier measurement or task changed." }, { "revision": "v1.2.8", "date": "2026-09-21", "note": "Added requested systems on the unchanged frozen 534-decision set, each through its author's own server and the existing TypeSafe adapter, one request at a time: decider-35b-a3b and reflex-27b (issues #4, #5), decider-2b (#2), reflex 4B (#3), OpenDecision, jev-local and LitJev on our RunPod GPUs; GLiNER2 large on our CPU; and decision-machine-1 (#8), a closed decision model behind milliseconds.ai's production API, shown in its own class. jqv (#6, #9) was re-run in full on our own GPU from its now-public serving code; that complete run replaces the v1.2.7 partial row. Bespoke Nimble 9B was re-run at Bespoke Labs' request after they raised its serving prompt limit from 2,048 to 8,192 tokens; the complete re-run replaces the v1.1.3 row (its old score is kept under superseded_rows). Mappings, endpoint conditions and cost bases were pushed before the runs (docs/v1.2-additions-run4.md, docs/v1.2-additions-run4b.md). No earlier measurement changed." }, { "revision": "v1.2.7", "date": "2026-09-20", "note": "Added three systems: jqv (a stock Qwen3-32B read as a decision model, submitted with a public endpoint) and the GLiNER2.5 small and multi checkpoints. The GLiNER2.5 rows ran the full frozen 534-decision set on our CPU with the same mapping as the GLiNER2 row. jqv is a partial row: its endpoint is the submitter's own machine, and this revision stopped sending held-out items to an endpoint a submitter operates. The easy and standard/judge tiers had already been sent in full when that was decided; the 109 held-out hard items never were, so the row covers 425 of 534 decisions and carries no rank. Mappings, endpoint conditions and cost bases were committed before any row was aggregated and before the published GLiNER2.5 runs started (docs/v1.2-additions-run3.md); jqv's run had begun about ten minutes earlier, but it needs no mapping and is priced at its base model's public tariff. No earlier row changed." }, { "revision": "v1.2.6", "date": "2026-09-20", "note": "Added openJev Verdict 1.4 and the identified SimpleJev public-demo configurations on the unchanged frozen 534-decision set. No earlier row changed." }, { "revision": "v1.2.5", "date": "2026-09-20", "note": "Added kev 0.5B and the 0.6B, 4B and 8B research previews. Each ran the full frozen v1.2 set (534 decisions including held-out items) through kev's native TypeSafe-compatible endpoint on an RTX 3090. No other row changed." }, { "revision": "v1.2.4", "date": "2026-09-20", "note": "classifier.dev (fast tier) leaves the ranking and becomes an honorable mention. It is not its own model: its own pages say \"The fast tier is Jev, TypeSafe's decision model\" (https://classifier.dev/benchmark), so ranking it against Jev ranks Jev's model against Jev's model at a different price. General rule from this revision on: a service that runs another entrant's model is listed with all of its scores and axes, but is not ranked against the models. Its numbers, axes, cost basis, radars and per-task outcomes are unchanged; only its rank is gone. Every other row moves up one place; no score changed." }, { "revision": "v1.2.3", "date": "2026-09-20", "note": "Cost correction. Every row's $ per 1,000 decisions is recomputed with each of the 534 decisions counted exactly once and priced exactly once. Three arithmetic mistakes were fixed: the 242-decision standard+judge run was averaged twice in the v1.1-tier price (556 rows instead of 314); rows priced from the gemini-3.1-flash-lite token counts used that run's standard+judge-only average (452 input tokens per decision) for all 314 v1.1 decisions instead of its average over all 314 (383); and requests whose answer came back unparseable were left unpriced although they were billed (9 DeepSeek V4.1 Flash decisions). The first two made the affected rows look 1.5-11 % more expensive than they are; the third made DeepSeek look 2.6 % cheaper. No tariff, no measurement, no item and no answer changed, and no rank changed. Details: results/v1.2/cost-correction-v1.2.3.json." }, { "revision": "v1.2.2", "date": "2026-09-19", "note": "Added five systems requested by readers: Laya, jeff, GLiNER2, openJev Verdict and classifier.dev (fast tier). Full v1.2 set each (534 decisions incl. held-out), scored with the unchanged v1.2 rules. Local systems ran on our CPU (4 threads) with the usual self-hosted latency adjustment; classifier.dev is a production API. Mappings were fixed before the runs (docs/v1.2-additions.md). No other row changed." }, { "revision": "v1.2.1", "date": "2026-09-19", "note": "Added djev (Maisa, diffusion-gemma): full v1.2 set (534 decisions incl. held-out) through its production API, scored with the unchanged v1.2 rules. Cost at djev's announced price ($0.035/M input tokens, output free), which is not yet charged (free preview). No other row changed." }, { "revision": "v1.2", "date": "2026-09-19", "note": "Final JevBench Score: 4 axes, geometric mean." } ], "protocol": "jevbench::v1.2", "status": "final", "generated_utc": "2026-09-21T22:55:40+00:00", "measured_in": "v1.2-wip (tag v1.2-wip); no measurement changed for v1.2 final; later additions measured on the same frozen items: v1.2.1: djev (Maisa, diffusion-gemma); v1.2.10: smalljev semantic-v9; v1.2.12: djev (thinking); v1.2.13: OpenJev (thinking, BF16); v1.2.14: Winnow-12B Q8; v1.2.15: Open-Jev 2B (Zefan Cai), Open-Jev 9B (Zefan Cai); v1.2.16: BAAI bge-reranker-v2-m3, Alibaba GTE Reranker ModernBERT-base, Mixedbread mxbai-rerank-base-v2, Qwen3-Reranker-4B, ZeroEntropy zerank-2; v1.2.2: classifier.dev (fast tier), GLiNER2 (Fastino, gliner2.5-base), jeff (Logan Markewich, GLiFormer 400M), Laya (Convai Innovations, ModernBERT-large 421M), openJev Verdict (heman10x, ModernBERT-base 151M); v1.2.5: kev 0.5B, kev 0.6B (research preview), kev 4B (research preview), kev 8B (research preview); v1.2.6: openJev Verdict 1.4, SimpleJev Qwen3.6-35B-A3B, SimpleJev Qwen3.8-27B; v1.2.7: GLiNER2.5 multi (Fastino, 287M), GLiNER2.5 small (Fastino, 74M); v1.2.8: decider-2b (Mapika), decider-35b-a3b (Mapika), decision-machine-1 (milliseconds.ai), GLiNER2 large (Fastino), jev-local (Qwen3.5-9B), jqv (Qwen3-32B zero-shot), LitJev (Qwen3.8-27B), Bespoke Nimble 9B (Bespoke Labs), OpenDecision (ModernBERT-large zero-shot), reflex-27b (Qwen3.8-27B), reflex 4B (kshetrajna12); v1.2.9: Certo v1 (AltSlate Labs)", "revision_note": "v1.3.0 scoring-only release: Intelligence is chance-corrected per tier and scores below 50 receive the growing near-chance penalty. Calibration, Speed, Cost, ranking eligibility, tasks and measurements are unchanged.", "score_name": "JevBench Score", "score_one_liner": "Intelligence above chance, Calibration, Speed, Cost — 25 % each, geometric mean; below 50 Intelligence receives a growing near-chance penalty.", "tiers": { "easy": 72, "judge": 146, "standard": 96, "hard": 220 }, "tier_weights": { "easy": 0.14, "standard": 0.28, "judge": 0.28, "hard": 0.3 }, "axis_weights": { "intelligence": 0.25, "calibration": 0.25, "speed": 0.25, "cost": 0.25 }, "presets": { "JevBench Score (25:25:25:25)": { "intelligence": 0.25, "calibration": 0.25, "speed": 0.25, "cost": 0.25 }, "Balanced 33:33:33 (no calibration)": { "intelligence": 0.3333333333333333, "calibration": 0.0, "speed": 0.3333333333333333, "cost": 0.3333333333333333 }, "Emphasis on Accuracy 60:20:20": { "intelligence": 0.6, "calibration": 0.0, "speed": 0.2, "cost": 0.2 }, "Emphasis on Speed 20:60:20": { "intelligence": 0.2, "calibration": 0.0, "speed": 0.6, "cost": 0.2 }, "Emphasis on Cost 20:20:60": { "intelligence": 0.2, "calibration": 0.0, "speed": 0.2, "cost": 0.6 }, "Intelligence only": { "intelligence": 1.0, "calibration": 0.0, "speed": 0.0, "cost": 0.0 } }, "main": "JevBench Score (25:25:25:25)", "speed_note": "Latency of self-hosted and demo endpoints is adjusted ×2 (+0.15 s on our own servers) to approximate production load — an assumption, not a measurement; raw measurements are in the table and the repo.", "cost_unit": { "unit": "$ per 1,000 decisions", "not_unit": "$ per 1,000 tokens", "one_liner": "Dollars per 1,000 decisions, not per 1,000 tokens: one decision is a whole question — state, rubric and options.", "worked_example": "One decision is a whole question, not a token. Jev 1.13.0 reads 950 input tokens per decision on average over the 534 v1.2 decisions. At its public tariff of $0.042 per MILLION input tokens (output tokens are free, https://docs.typesafe.ai/models), 1,000 decisions therefore cost 950 x 1,000 x $0.042 / 1,000,000 = $0.0399. That is what the Cost column shows: $0.0399 per 1,000 decisions, not per 1,000 tokens.", "short_note": "One decision ≈ 950 input tokens on average; at Jev's $0.042 per million input tokens that is $0.0399 per 1,000 decisions.", "mean_input_tokens_per_decision_jev": 950.3389513108614 }, "cost_correction": { "revision": "v1.2.3", "file": "results/v1.2/cost-correction-v1.2.3.json", "what_was_wrong": [ "The v1.1 and v1.1.3 aggregations built their cost average from a row list that contained the 242-decision standard+judge run twice (once as the standard tier, once as the judge tier) and the 72 easy decisions once: 556 rows instead of 314. The standard and judge tiers were therefore over-weighted in the price, which made the affected rows look 1.5-3.3 % more expensive than they are.", "Rows without their own token counts were priced at the input tokens of the gemini-3.1-flash-lite run measured on the 242 standard+judge decisions only (452 per decision) and that figure was applied to all 314 v1.1 decisions, which excludes the shorter easy tier. Over all 314 decisions the same run averages 383.41 input tokens, which is the figure used from v1.2.3 on. This made the affected rows look 4-11 % more expensive.", "A metered row's price left out the requests whose answer came back unparseable. Those requests returned HTTP 200 with generated tokens and were billed, and JevBench already counts them as wrong answers, so from v1.2.3 they are priced too. Only DeepSeek V4.1 Flash had any (9 of its 314 v1.1 decisions); its price rises by 2.6 %.", "No tariff was wrong. The hard-tier costs, and classifier.dev's flat plan price, were already correct." ], "rule": "usd_per_1000_v11_tiers = 1000 x (mean input tokens per decision x $/M in + output tokens charged x $/M out) / 1e6, over all 314 v1.1 decisions (72 easy + 242 standard+judge), each decision counted exactly once and priced exactly once. A metered row uses the provider's own tariff and its own measured token counts, including the requests whose answer could not be parsed; an estimated row uses the reference tariff for its weights or size class and, when the run reports no usage, the input tokens of the gemini-3.1-flash-lite run on the same prompts over the same 314 decisions. usd_per_1000 = (v11 x 314 + hard x 220) / 534." }, "cost_correction_table": { "classifier-dev-fast": { "old": 0.003333333333333333, "new": 0.003333333333333333, "pct": 0.0, "unchanged": true }, "jev-1.13.0": { "old": 0.04061412193840432, "new": 0.03991423595505617, "pct": -1.7232576994021993, "unchanged": false }, "semif-qwen3.5-4b": { "old": 0.022975040619190038, "new": 0.02244460674157303, "pct": -2.3087396727993554, "unchanged": false }, "djev": { "old": 0.025951254681647943, "new": 0.025951254681647943, "pct": 0.0, "unchanged": true }, "laya": { "old": 0.0028831273408239703, "new": 0.0028831273408239703, "pct": 0.0, "unchanged": true }, "open-alternative-jev": { "old": 0.022171404494382024, "new": 0.022171404494382024, "pct": 0.0, "unchanged": true }, "system-one-open": { "old": 0.015685659145076917, "new": 0.014880936329588014, "pct": -5.130309208213748, "unchanged": false }, "openjev-razorback16": { "old": 0.0671907566081966, "new": 0.06560455056179774, "pct": -2.3607503866169424, "unchanged": false }, "jeff": { "old": 0.006036722846441948, "new": 0.006036722846441948, "pct": 0.0, "unchanged": true }, "openjev-sglang": { "old": 0.134615554522674, "new": 0.13127546816479402, "pct": -2.481203877013638, "unchanged": false }, "openjev-verdict": { "old": 0.003872921348314607, "new": 0.00367125468164794, "pct": -5.20709429729129, "unchanged": false }, "gpt-5.6-luna": { "old": 0.24728613154420276, "new": 0.24191123595505612, "pct": -2.173553185365702, "unchanged": false }, "open-jev-deberta-v3-large": { "old": 0.007742829572538459, "new": 0.007340468164794008, "pct": -5.196568050154547, "unchanged": false }, "nimble-9b": { "old": 0.10850016283992836, "new": 0.10494152501680593, "pct": -3.2798456057365897, "unchanged": false }, "gemini-3.1-flash-lite": { "old": 0.26820170492819223, "new": 0.26378698501872655, "pct": -1.6460446851550288, "unchanged": false }, "deepseek-flash": { "old": 0.5788175142273461, "new": 0.593682584269663, "pct": 2.5681790334489274, "unchanged": false }, "system-one-sg": { "old": 0.09153429437798077, "new": 0.08944116853932584, "pct": -2.286712158408745, "unchanged": false }, "gliner2": { "old": 0.003872921348314607, "new": 0.00367125468164794, "pct": -5.20709429729129, "unchanged": false }, "qwen3.8-27b": { "old": 2.7110372815546095, "new": 2.669088141927965, "pct": -1.5473464681603164, "unchanged": false }, "needle-3-tools": { "old": 0.016219537190082647, "new": 0.014372006369426754, "pct": -11.3907739721794, "unchanged": false }, "needle-3": { "old": 0.02492563984585384, "new": 0.023839264044943822, "pct": -4.358467054921875, "unchanged": false } }, "scoring": { "jevbench_score": "Geometric mean of Intelligence, Calibration, Speed and Cost, 25 % each. If chance-corrected Intelligence is below 50, multiply by (Intelligence / 50)^2; at or above 50 there is no penalty.", "intelligence": "Per tier: 100 x (accuracy - chance) / (1 - chance), clipped at 0. Chance is 1 / options for each item (1 / levels for score items), then averaged within the tier. Tier weights: hard 30 %, easy 14 %, standard 28 %, judge 28 %. Failed, timed-out or unparseable answers count as wrong.", "hard_tier": "220 new decisions (111 public, 109 held out): long multi-condition policy documents (2-6k tokens), priority trade-offs, deliberately ambiguous cases with a 'no clear answer' label, traps, multi-hop lookups, date/number reasoning, adversarial distractors, subtle answer-judging, overlapping routing, and probability items with an exact gold distribution. Half written by Claude Opus 5, half by GPT-5.6 Sol; each item reviewed blind and then against its gold by the other model; one discussion round; frozen and hashed before any benchmarked system saw an item. No item was selected on any system's answers.", "calibration": "Hard tier only, systems that return a probability distribution: mean of (a) 100 x (1 - ECE/0.5), ECE = top-label expected calibration error in 10 bins, and (b) probability fidelity = 100 x (1 - mean total-variation distance) between the returned distribution and the exact gold distribution on the 20 probability items. Label-only systems have none; it counts as 0 in the JevBench Score.", "speed": "Mean of score(p50) and score(p95) of the serial 242-decision standard+judge run; score(s) = 100 - 20 log10(s / 0.1 s), clipped to 0..100 (0.1 s = 100, 1 s = 80, 10 s = 60). Latency of self-hosted and demo endpoints is adjusted ×2 (+0.15 s on our own servers) to approximate production load — an assumption, not a measurement; raw measurements are in the table and the repo. Production APIs (Jev, djev, classifier.dev, OpenAI, Google, DeepSeek, Chutes) are not adjusted.", "cost": "US dollars per 1,000 DECISIONS — not per 1,000 tokens. One decision is one whole question: its state, its rubric and its options, which is hundreds to thousands of input tokens. Pooled over all 534 v1.2 decisions; score = 100 - 30 log10(usd / 0.001), clipped to 0..100 ($0.001 = 100, $0.01 = 70, $0.10 = 40, $1 = 10). Measured = public tariff x measured tokens. est. = hosted-provider list price of the same weights or size class x tokens (for a flat-rate service, its published plan price at full use). announced = the provider's published price, not yet charged (free preview), x measured tokens.", "ranked": "Ranked: a system's own model, with every tier attempted for >= 95 % of its decisions. Partial runs are shown below the ranking, marked, without a rank. A service that runs another entrant's model is listed with all of its scores and axes, but is not ranked against the models. Ranking it would rank the same model twice, once at the model's own price and once at the service's. The row keeps every number, axis, cost basis and per-task outcome; it carries no rank number.", "presets": "Other views reweight the same four axes and combine them the same way (geometric mean). They are not the JevBench Score." }, "hard_dataset": { "frozen_utc": "2026-09-19T11:43:54+00:00", "n_items": 220, "n_public": 111, "n_heldout": 109, "families": { "adversarial": 12, "ambiguous": 14, "judge_hard": 33, "long_policy": 38, "multi_hop": 35, "probability": 20, "routing_hard": 10, "temporal_numeric": 30, "tradeoff": 12, "trap": 16 }, "types": { "choice": 129, "score": 14, "noul": 77 }, "authors": { "gpt-5.6-sol": 110, "claude-opus-5": 110 }, "sha256_public_file": "89e9e6becb33ed88c1de7d42dcc87531b2fb64cfaef4e1986faf7c37b3f80ebb", "sha256_heldout_file": "4e8adf72988766534c87c2b83808bde7f0f934f515f8eeb01cc74eb90b8778b1", "dataset_hash_all": "ec200ccd3db28153c93bfeaed55acb18b4909610ba403cf73482abbc4074ef6b", "review": { "opus-a": { "authored": 40, "accepted": 40, "rejected": [], "no_verdict": [], "source_file": "authoring/opus-a.jsonl" }, "opus-b": { "authored": 40, "accepted": 40, "rejected": [], "no_verdict": [], "source_file": "authoring/opus-b.jsonl" }, "sol-a": { "authored": 40, "accepted": 40, "rejected": [], "no_verdict": [], "source_file": "authoring/sol-a.v2.jsonl" }, "sol-b": { "authored": 40, "accepted": 40, "rejected": [], "no_verdict": [], "source_file": "authoring/sol-b.v2.jsonl" }, "opus-c": { "authored": 30, "accepted": 30, "rejected": [], "no_verdict": [], "source_file": "authoring/opus-c.jsonl" }, "sol-c": { "authored": 30, "accepted": 30, "rejected": [], "no_verdict": [], "source_file": "authoring/sol-c.v3.jsonl" } }, "rule": "Items authored by Claude Opus 5 were reviewed by GPT-5.6 Sol and vice versa (blind answer, then gold verdict); one discussion round; anything not accepted afterwards was dropped. No item was selected or dropped on the basis of any benchmarked system's answers. Frozen before any benchmarked system saw an item." }, "footnotes": { "open-alternative-jev": "With the options in reverse order (A. no, B. yes) the same model scored 21 % instead of 72 % on yes/no answer-judging items — small models are very sensitive to option order.", "bge-reranker-v2-m3": "Neutral documented reranker adapter; instruction and no-instruction public calibration were run, then frozen before one held-out pass.", "certo": "The public Certo v1 checkpoint through the author's DecisionModel, serially on our rented GPU. The question instruction is prepended to the state because Certo exposes state + runtime options but no separate question field; the published 64-token state and 48-token option limits are unchanged. The model card says v1 does not yet transfer to arbitrary natural-language prose. Cost is an estimate from same-size hosted encoders times the checkpoint's retained input tokens, not free/100.", "classifier-dev-fast": "Its own benchmark page says the fast tier is Jev. Free for us; the price is its published Pro plan ($20/month for 200,000 fast classifications a day) at full use, $0.0033 per 1,000 decisions.", "decider-2b": "The author's TypeSafe-compatible server and published weights (Qwen3.5-2B-Base with a trained one-pass decision readout), run serially on our GPU. Self-host latency gets the standard ×2 + 0.15 s adjustment.", "decider-35b-a3b": "The author's TypeSafe-compatible server and published FP8 weights, run serially on our H100 NVL. The exhaustive startup batch warmup was skipped; each required serial shape captured lazily before its measured request. Self-host latency receives the standard ×2 + 0.15 s adjustment. Cost uses the closest hosted 35B-A3B input tariff and is not the temporary rental charge.", "decision-machine-1": "A closed-weights decision model behind a production API that serves TypeSafe's wire format, so the unchanged typesafe adapter ran it. Run on a free test key (30 requests a minute, 2.2 s between requests); the provider states the inference infrastructure is the same as for paid keys. Cost is the public paid tariff, $0.04 per million input tokens (output free), times the input tokens the API reported.", "djev-thinking": "Experimental full-generation path over the same DiffusionGemma checkpoint as djev-dev: thinking was enabled and the model could generate up to 8,192 tokens before returning its distribution. Current djev-dev itself hard-codes enable_thinking=false, diffusion_max_steps=1 and read_only=true, so this is not a switch in its published typed API. It is substantially slower/costlier, and 72/534 requests exhausted the output budget without a parseable distribution; those are failures. Cost uses measured tokens and a same-size hosted reference, not the H200 rental bill.", "djev": "The measured endpoint was Maisa's hosted API in free preview; the cost uses its announced price ($0.035 per million input tokens, output free), and nothing was charged. The self-hostable djev-dev runtime is Apache-2.0 and applies a structured one-step inference method to Google's Apache-2.0 diffusiongemma-26B-A4B-it checkpoint; it adds no separately trained djev weights. Probabilities are djev's own (its docs call them experimental and uncalibrated).", "gliner2-large": "The large checkpoint of Fastino's earlier GLiNER2 family, same documented mapping as the GLiNER2 row: the question goes in front of the text and the probabilities are the model's own single-label softmax over the labels, read out in full. A general schema classifier, not a Jev rebuild.", "gliner2.5-multi": "The multilingual GLiNER2.5 checkpoint (287M), same family and same documented mapping as the GLiNER2 row. JevBench items are English only, so its multilingual training is not exercised here.", "gliner2.5-small": "The small GLiNER2.5 checkpoint (74M), same family and same documented mapping as the GLiNER2 row: the question goes in front of the text and the probabilities are the model's own single-label softmax over the labels, read out in full. A general schema classifier, not a Jev rebuild.", "gliner2": "A general schema classifier, not a Jev rebuild. The question goes in front of the text; the probabilities are GLiNER2's own single-label softmax over the labels, read out in full (mapping fixed before the run).", "gte-reranker-modernbert-base": "Neutral documented reranker adapter; instruction and no-instruction public calibration were run, then frozen before one held-out pass.", "jeff": "Self-hosted from its GitHub repo with server defaults, on our CPU (the author recommends a GPU, e.g. an L4), through the same TypeSafe-compatible API as Jev.", "jev-local": "The author's local Jev-compatible server in its default full configuration: a frozen Qwen3.5-9B scores each option by its mean log-probability (one forward pass per option, no generation, no decision training). Run serially on our GPU. It re-reads the state once per option; if its reported token count covers one pass only, a per-token hosted price would be higher than this estimate.", "jqv": "A stock Qwen3-32B with no decision training: the state is prefilled once, each question is an isolated branch and the answer is read from the option-letter logits, with one fitted temperature (3.02, 400 MMLU validation items). Re-run in v1.2.8 on our own GPU from the now-public serving code (Octalab-Inc/jqv 0189b67), so all 534 decisions including the held-out hard items were asked; this full run replaces the v1.2.7 partial row, which had been measured on the submitter's machine. Cost is the base model's public per-token tariff, not free.", "kev-0.5b": "Self-hosted from the author's repository at commit 20fa626 through its native TypeSafe-compatible `/v1/systemone` server, BF16 on an RTX 3090; measured serially from Sandy over the internet. This is the v0.1 release.", "kev-0.6b": "Self-hosted from the author's repository at commit 20fa626 through its native TypeSafe-compatible `/v1/systemone` server, BF16 on an RTX 3090; measured serially from Sandy over the internet. The author labels this checkpoint a research preview.", "kev-4b": "Self-hosted from the author's repository at commit 20fa626 through its native TypeSafe-compatible `/v1/systemone` server, BF16 on an RTX 3090; measured serially from Sandy over the internet. The author labels this checkpoint a research preview.", "kev-8b": "Self-hosted from the author's repository at commit 20fa626 through its native TypeSafe-compatible `/v1/systemone` server, BF16 on an RTX 3090; measured serially from Sandy over the internet. The author labels this checkpoint a research preview.", "laya": "The English checkpoint (repo root), run on our CPU through its own `laya` package. Its budget is 512 tokens per question, so long hard-tier states are cut by the package itself.", "litjev": "The author's reproduction of Jev's decision layer on an off-the-shelf model, in its default configuration: Qwen3.8-27B, scores read from the output head, no training and no calibration file (its README says probabilities are not calibrated by default). Run serially on our GPU through an SSH tunnel, because its server binds to localhost; the request still crosses the internet and gets the ×2 + 0.15 s adjustment.", "mxbai-rerank-base-v2": "Neutral documented reranker adapter; instruction and no-instruction public calibration were run, then frozen before one held-out pass.", "nimble-9b": "Re-run in v1.2.8 at Bespoke Labs' request after they raised the serving prompt limit from 2,048 to 8,192 tokens (bespokelabsai/nimble PR #4). Same recipe as the v1.1.3 run — the published LoRA merged into Qwen3.5-9B with the author's PEFT safe-merge, served with SGLang and the author's Jev-compatible API — now from current nimble main; the adapter weights are unchanged. Hard-tier accuracy rose from 43.6 % to 65.5 %, yet the score fell: the long hard items that used to fail at once are now answered and priced (so Cost fell), and this pod was in Canada while the v1.1.3 run's was in Sweden, so part of the lower Speed is network distance from our server in Germany. This complete run replaces the earlier row; its old score is kept in the artifact under superseded_rows.", "open-jev-zefan-2b": "The author's pinned LoRA adapter, trained scalar decision head and calibration temperature, served by the author's Open-Jev server with prefix caching off, batch size 1 and 4,096-token limit. Serial requests were measured from Sandy over an SSH tunnel to the H100. Self-host latency receives the standing x2 + 0.15 s adjustment. Cost uses the exact Qwen3.5-9B hosted input tariff for 9B and the same conservative same-family proxy for the unlisted 2B; neither receives an automatic 100. Exact normalized comparison found no JevBench public task state or instruction in the 79,116-row public training projection.", "open-jev-zefan-9b": "The author's pinned LoRA adapter, trained scalar decision head and calibration temperature, served by the author's Open-Jev server with prefix caching off, batch size 1 and 4,096-token limit. Serial requests were measured from Sandy over an SSH tunnel to the H100. Self-host latency receives the standing x2 + 0.15 s adjustment. Cost uses the exact Qwen3.5-9B hosted input tariff for 9B and the same conservative same-family proxy for the unlisted 2B; neither receives an automatic 100. Exact normalized comparison found no JevBench public task state or instruction in the 79,116-row public training projection.", "opendecision": "A zero-shot NLI classifier behind a TypeSafe-compatible server, not a trained decision model: it scores each option as an entailment hypothesis with ModernBERT-large-zeroshot-v2.0. Its choice path runs several NLI passes over the same state, which the reported token count does not include, so a per-token hosted price would be higher than the estimate here. Pre-registered for our CPU in v1.2.7, run on our GPU because the CPU was far too slow.", "openjev-thinking": "OpenJev's real typed-API thinking switch at think=512, using its own /v1/systemone server over BF16 DiffusionGemma. The thought is generated first, then native probability reads are taken after it. All 534 requests returned valid distributions. Cost counts the server's billed input and thought output tokens.", "openjev-verdict-1.4": "Same public weights as the earlier Verdict row, run through the author's fixed v1.4 engine. That engine auto-loads the calibrator for every option count, frames candidate labels as NLI sentences and uses a 512-token context budget. Run locally on our CPU, serially.", "openjev-verdict": "The openJev-verdict-2.0 Hugging Face repo ships no weights; its config is byte-identical to heman10x/rlcd-modernbert-151m, whose published weights we ran with the author's engine. The 'verdict2-base' checkpoint behind the README's numbers is not downloadable yet (Git LFS 404); we will run it once it is.", "qwen3-reranker-4b": "Neutral documented reranker adapter; instruction and no-instruction public calibration were run, then frozen before one held-out pass.", "reflex-27b": "The frozen public Qwen3.8-27B checkpoint through reflex at the requested pinned commit, with two option orders averaged and temperature 1. No adapter or fitted calibration file. Run serially on our H100 NVL. Self-host latency receives the standard ×2 + 0.15 s adjustment; cost uses the exact base model's public hosted input tariff.", "reflex-4b": "The author's reflex-serve: Qwen3.5-4B with the published LoRA and its per-primitive calibration file; the state is encoded once and each question read from the label logits. Run serially on our GPU; the author discloses that the 231 public items were used four times as a development gate.", "simplejev-qwen3.6-35b-a3b": "Author's no-login shared demo, model id recorded verbatim, one request at a time at or below its 2 RPS limit. SimpleJev reads answer-token logits and returns the complete distribution; it does not generate an answer. Speed uses the public-demo x2 load adjustment; cost uses a hosted size-class input price and is not free/100.", "simplejev-qwen3.8-27b": "Author's no-login shared demo, model id recorded verbatim, one request at a time at or below its 2 RPS limit. SimpleJev reads answer-token logits and returns the complete distribution; it does not generate an answer. Speed uses the public-demo x2 load adjustment; cost uses a hosted size-class input price and is not free/100.", "smalljev": "The public semantic-v9 LoRA and native heads over MiniCPM5-2B-Base, through the mapping frozen before the run. It has a typed Python contract but no TypeSafe-compatible HTTP route. The released training recipe explicitly hill-climbed against JevBench's public shape and source families; this allowed public benchmark-directed development is disclosed. Cost is $0.04/M measured input tokens, not free/100.", "winnow-12b": "The submitted Q8_0 GGUF ran through the pinned author's TypeSafe-compatible /v1/systemone server with 8,192 context, four resident decision branches, Q8 KV, and full GPU offload. The private training corpus was not released. The author's checksum-based audit reports zero exact public-item overlap, but that claim cannot be independently reproduced; our scan found no exact public state or instruction text in the released artifacts. Cost uses the $0.05/M-input hosted Gemma 3 12B reference, not free/100.", "zerank-2": "Neutral documented reranker adapter; instruction and no-instruction public calibration were run, then frozen before one held-out pass." }, "superseded_rows": { "nimble-9b": { "reason": "Re-run in v1.2.8 after Bespoke Labs raised the serving prompt limit from 2,048 to 8,192 tokens (bespokelabsai/nimble PR #4); the v1.1.3 run had failed long items on that limit.", "old_score": 61.45878169849573, "old_tiers": { "easy": 1.0, "standard": 0.9479166666666666, "judge": 0.8904109589041096, "hard": 0.43636363636363634 }, "old_endpoint_condition": "our RunPod GPU (A40 48 GB (EU-SE-1)), reached over the internet" } }, "excluded_runs": [ { "key": "open-alternative-jev-reversed-order", "run_key": "open-alternative-jev", "why_not_ranked": "Our first adapter put the options in reverse order (A. no, B. yes); the author's yes_no() helper builds A. yes, B. no. An adapter mistake, not a model weakness, so the author-order run is the ranked row. Raw run files are kept.", "tiers": { "easy": 1.0, "standard": 0.8125, "judge": 0.5068493150684932, "hard": 0.55 }, "footnote": "With the options in reverse order (A. no, B. yes) the same model scored 21 % instead of 72 % on yes/no answer-judging items — small models are very sensitive to option order." } ], "honorable_mentions": { "heading": "Honorable mentions — services built on another entrant's model", "rule": "A service that runs another entrant's model is listed with all of its scores and axes, but is not ranked against the models. Ranking it would rank the same model twice, once at the model's own price and once at the service's. The row keeps every number, axis, cost basis and per-task outcome; it carries no rank number.", "systems": { "classifier-dev-fast": { "runs_on_key": "jev-1.13.0", "runs_on": "Jev (TypeSafe)", "short_reason": "runs on Jev (TypeSafe) — listed, not ranked", "why_not_ranked": "classifier.dev is not its own model. Its own pages say so: \"The fast tier is Jev, TypeSafe's decision model\" (https://classifier.dev/benchmark, read 2026-09-20), and the API answers with \"model\": \"jev-1.13.0\" — the same model version this benchmark measures directly as Jev 1.13.0. What it adds is a price and, on its smart tier, an orchestration layer: \"The smart tier is Jev plus a reasoning model re-asking only the answers Jev put under 0.7 confidence\" — escalation on low confidence (a model cascade), not best-of-N, not self-consistency and not a committee. Its published escalation model is gemini-3.8-flash. Ranking it against Jev would rank Jev's model against Jev's model, so from v1.2.4 it is an honorable mention instead of #1.", "tier_measured": "Only the fast tier was measured. The smart tier's escalation was never run, so nothing here scores it.", "price_note": "$0.0033 per 1,000 decisions is an estimate from the published flat-rate plan at full use: classifier.dev Pro is $20/month for 200,000 fast classifications a day (https://classifier.dev/pricing, read 2026-09-20), and one classification is one decision. Lower use costs more per decision — at a tenth of that allowance it is $0.033 per 1,000 — and the free tier (20,000 fast classifications a day), which is what our run used, costs nothing. Their pages do not say how the flat rate is funded, so we do not know their cost basis; the only figure they publish is what the model costs a caller: \"The model behind the fast tier costs about $0.005 per thousand classifications and needs a TypeSafe key\" (https://classifier.dev/pricing) — for their short single-sentence inputs, not for JevBench's whole questions.", "not_pass_through": "On our set the fast tier scored 97.3 % on the judge tier against Jev's 94.5 %, and 70.5 % against 74.1 % on the hard tier. classifier.dev's own explanation for differences of this kind is batching (\"The fast tier is Jev, packed a thousand to a request\"); on their own two test sets they measured the same difference as noise.", "credit": "A legitimate, well-documented product: free without an account, open source (https://github.com/mrmps/classifier-dev), by Michael Ryaboy (@michael_chomsky).", "sources": [ "https://classifier.dev", "https://classifier.dev/benchmark", "https://classifier.dev/pricing", "https://classifier.dev/about" ], "sources_read": "2026-09-20" } } }, "systems": [ { "key": "jev-1.13.0", "display": "Jev 1.13.0 (TypeSafe AI)", "class": "jev", "open": "no", "author": "TypeSafe AI", "repo": "https://docs.typesafe.ai", "licence": "proprietary API", "underlying": "closed", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "production API (api.typesafe.ai)", "endpoint_kind": "api", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.9895833333333334, "judge": 0.9452054794520548, "hard": 0.740909090909091 }, "axes": { "intelligence": 85.69176285262259, "calibration": 82.6528888888889, "speed": 83.26811926100174, "cost": 51.96616538951724 }, "jevbench_score": 74.40448849535433, "speed": { "p50_s_raw": 0.6524335257709026, "p95_s_raw": 0.7221905551850795, "p50_s_adjusted": 0.6524335257709026, "p95_s_adjusted": 0.7221905551850795, "adjustment": "none (production API)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.6717139892280102, "hard_tier_p95_s": 0.8295219600200652 }, "cost": { "kind": "measured", "usd_per_1000": 0.03991423595505617, "basis": "public tariff x measured tokens (https://docs.typesafe.ai/models (output tokens not billed)) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | public tariff x measured tokens (hard-tier run)", "usd_per_1000_v11_tiers": 0.024698140127388538, "usd_per_1000_hard": 0.06163175454545452, "self_host_sensitivity": null }, "calibration": { "score": 82.6528888888889, "score_label_only_as_onehot": null, "ece_hard": 0.06061111111111118, "probability_fidelity": 77.42800000000001, "brier_hard": 0.339603665766944, "brier_standard_judge_v11": 0.05558677685950414, "note": null }, "hard": { "run": "runs/jev-1.13.0--hard", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.740909090909091, "by_family": { "adversarial": { "correct": 12, "n": 12, "accuracy": 1.0 }, "ambiguous": { "correct": 11, "n": 14, "accuracy": 0.7857142857142857 }, "judge_hard": { "correct": 26, "n": 33, "accuracy": 0.7878787878787878 }, "long_policy": { "correct": 23, "n": 38, "accuracy": 0.6052631578947368 }, "multi_hop": { "correct": 30, "n": 35, "accuracy": 0.8571428571428571 }, "probability": { "correct": 16, "n": 20, "accuracy": 0.8 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 8, "n": 30, "accuracy": 0.26666666666666666 }, "tradeoff": { "correct": 11, "n": 12, "accuracy": 0.9166666666666666 }, "trap": { "correct": 16, "n": 16, "accuracy": 1.0 } }, "has_distribution": true, "brier_mean": 0.339603665766944, "ece": 0.06061111111111118, "probability_fidelity": 77.42800000000001, "calibration_score": 82.6528888888889, "onehot": { "ece": 0.25909090909090904, "probability_fidelity": 59.264999999999986, "calibration_score": 53.72340909090909 }, "latency_p50_s": 0.6717139892280102, "latency_p95_s": 0.8295219600200652, "mean_input_tokens": 1467.4227272727273, "mean_output_tokens": 45.086363636363636, "charged_usd": 0.013558985999999993 }, "presets": { "JevBench Score (25:25:25:25)": 74.40448849535433, "Balanced 33:33:33 (no calibration)": 71.84217985288262, "Emphasis on Accuracy 60:20:20": 77.09093830757904, "Emphasis on Speed 20:60:20": 76.21127073034907, "Emphasis on Cost 20:20:60": 63.11258509600221, "Intelligence only": 85.69176285262262 }, "rank": 1, "rank_under": { "JevBench Score (25:25:25:25)": 1, "Balanced 33:33:33 (no calibration)": 3, "Emphasis on Accuracy 60:20:20": 2, "Emphasis on Speed 20:60:20": 4, "Emphasis on Cost 20:20:60": 9, "Intelligence only": 5 } }, { "key": "semif-qwen3.5-4b", "display": "SemIf, formerly OpenJev (Qwen3.5-4B, TheoLeeCJ)", "class": "jev-rebuild", "open": "yes", "author": "Theodore Lee (TheoLeeCJ)", "repo": "https://github.com/TheoLeeCJ/openjev", "licence": "MIT (code); Qwen3.5 weights Apache-2.0", "underlying": "Qwen/Qwen3.5-4B (frozen, BF16)", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our RunPod GPU (RTX PRO 4500 Blackwell 32 GB (EU-RO-1)), reached over the internet", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.9791666666666666, "judge": 0.952054794520548, "hard": 0.5954545454545455 }, "axes": { "intelligence": 78.9598081649456, "calibration": 72.60139737235289, "speed": 83.70422262345133, "cost": 59.46663998783557 }, "jevbench_score": 73.08747495585776, "speed": { "p50_s_raw": 0.19796114787459373, "p95_s_raw": 0.3153164997696876, "p50_s_adjusted": 0.5459222957491875, "p95_s_adjusted": 0.7806329995393753, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": "RunPod RTX PRO 4500 Blackwell 32 GB (EU-RO-1)", "measured_where": "from a Hetzner server in Germany over the internet to the pod's public TCP port (plain HTTP, one connection per request) through a thin transport around the author's library; model loaded before timing", "hard_tier_p50_s": 0.22315140068531036, "hard_tier_p95_s": 0.6500909611582756 }, "cost": { "kind": "estimate", "usd_per_1000": 0.02244460674157303, "basis": "ESTIMATE: hosted-provider price, deepinfra Qwen/Qwen3.5-4B list price $0.03/M in, $0.15/M out (same weights (not on OpenRouter), as open-alternative-jev in v1.1.2) x 396 input and 1 output tokens per decision (input tokens measured) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | ESTIMATE: deepinfra Qwen/Qwen3.5-4B $0.03/M in, $0.15/M out x 1244 in / 0 out tokens per hard decision", "usd_per_1000_v11_tiers": 0.012019777070063693, "usd_per_1000_hard": 0.03732368181818181, "self_host_sensitivity": { "usd_per_1000": 0.0347231924417574, "score": 61.4845088183062, "machine": "1x RTX PRO 4500 Blackwell 32 GB (EU-RO-1) (on-demand, RunPod secure)", "usd_per_h": 0.72, "concurrency": 4, "utilisation": 0.3, "p50_s_used": 0.2083391546505444, "decisions_per_hour": 20735.42060418796 } }, "calibration": { "score": 72.60139737235289, "score_label_only_as_onehot": null, "ece_hard": 0.12080109007622307, "probability_fidelity": 69.36301275995041, "brier_hard": 0.5419578429506192, "brier_standard_judge_v11": 0.06934070252416917, "note": null }, "hard": { "run": "runs-gpu/semif-qwen3.5-4b--hard", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.5954545454545455, "by_family": { "adversarial": { "correct": 10, "n": 12, "accuracy": 0.8333333333333334 }, "ambiguous": { "correct": 8, "n": 14, "accuracy": 0.5714285714285714 }, "judge_hard": { "correct": 26, "n": 33, "accuracy": 0.7878787878787878 }, "long_policy": { "correct": 16, "n": 38, "accuracy": 0.42105263157894735 }, "multi_hop": { "correct": 22, "n": 35, "accuracy": 0.6285714285714286 }, "probability": { "correct": 8, "n": 20, "accuracy": 0.4 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 6, "n": 30, "accuracy": 0.2 }, "tradeoff": { "correct": 9, "n": 12, "accuracy": 0.75 }, "trap": { "correct": 16, "n": 16, "accuracy": 1.0 } }, "has_distribution": true, "brier_mean": 0.5419578429506192, "ece": 0.12080109007622307, "probability_fidelity": 69.36301275995041, "calibration_score": 72.60139737235289, "onehot": { "ece": 0.40454545454545454, "probability_fidelity": 42.757, "calibration_score": 30.923954545454546 }, "latency_p50_s": 0.22315140068531036, "latency_p95_s": 0.6500909611582756, "mean_input_tokens": 1244.1227272727272, "mean_output_tokens": 0.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 73.08747495585776, "Balanced 33:33:33 (no calibration)": 73.25022293617522, "Emphasis on Accuracy 60:20:20": 75.48276568116796, "Emphasis on Speed 20:60:20": 77.26526837571738, "Emphasis on Cost 20:20:60": 67.38988739226828, "Intelligence only": 78.95980816494558 }, "rank": 2, "rank_under": { "JevBench Score (25:25:25:25)": 2, "Balanced 33:33:33 (no calibration)": 2, "Emphasis on Accuracy 60:20:20": 3, "Emphasis on Speed 20:60:20": 2, "Emphasis on Cost 20:20:60": 4, "Intelligence only": 18 } }, { "key": "djev", "display": "djev (Maisa, diffusion-gemma)", "class": "jev-rebuild", "open": "yes", "author": "Maisa (David Villalón)", "repo": "https://github.com/Davipar/djev-dev", "licence": "Apache-2.0 code; Google DiffusionGemma Apache-2.0 weights; no djev-specific weights", "underlying": "inference method on google/diffusiongemma-26B-A4B-it (one structured denoising read), not a separately trained model", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "production API (api.djev.dev, free preview)", "endpoint_kind": "api", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.9791666666666666, "judge": 0.9315068493150684, "hard": 0.6954545454545454 }, "axes": { "intelligence": 82.66796898754649, "calibration": 65.41450619010376, "speed": 91.35677310946093, "cost": 57.57524920597589 }, "jevbench_score": 73.02927314568292, "speed": { "p50_s_raw": 0.2370578795671463, "p95_s_raw": 0.30865143015980717, "p50_s_adjusted": 0.2370578795671463, "p95_s_adjusted": 0.30865143015980717, "adjustment": "none (production API)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.24707749113440514, "hard_tier_p95_s": 0.3622760258615016 }, "cost": { "kind": "announced", "usd_per_1000": 0.025951254681647943, "basis": "ANNOUNCED PRICE (free preview): djev's docs state $0.035 per million input tokens, output tokens free (https://api.djev.dev/docs, 'Usage & credits'; prepaid billing not yet switched on, 19 Sep 2026, so nothing was charged) x measured input tokens (741 per decision on average over all 534 decisions)", "usd_per_1000_v11_tiers": 0.013884410828025478, "usd_per_1000_hard": 0.04317393181818183, "self_host_sensitivity": null }, "calibration": { "score": 65.41450619010376, "score_label_only_as_onehot": null, "ece_hard": 0.17503581317955752, "probability_fidelity": 65.83617501611904, "brier_hard": 0.4680494813450917, "brier_standard_judge_v11": 0.08268269187152161, "note": null }, "hard": { "run": "runs/djev--hard (job djev-jevbench-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.6954545454545454, "by_family": { "adversarial": { "correct": 12, "n": 12, "accuracy": 1.0 }, "ambiguous": { "correct": 10, "n": 14, "accuracy": 0.7142857142857143 }, "judge_hard": { "correct": 29, "n": 33, "accuracy": 0.8787878787878788 }, "long_policy": { "correct": 18, "n": 38, "accuracy": 0.47368421052631576 }, "multi_hop": { "correct": 29, "n": 35, "accuracy": 0.8285714285714286 }, "probability": { "correct": 12, "n": 20, "accuracy": 0.6 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 10, "n": 30, "accuracy": 0.3333333333333333 }, "tradeoff": { "correct": 8, "n": 12, "accuracy": 0.6666666666666666 }, "trap": { "correct": 15, "n": 16, "accuracy": 0.9375 } }, "has_distribution": true, "brier_mean": 0.4680494813450917, "ece": 0.17503581317955752, "probability_fidelity": 65.83617501611904, "calibration_score": 65.41450619010376, "onehot": { "ece": 0.30454545454545456, "probability_fidelity": 52.93449999999999, "calibration_score": 46.01270454545454 }, "latency_p50_s": 0.24707749113440514, "latency_p95_s": 0.3622760258615016, "mean_input_tokens": 1233.540909090909, "mean_output_tokens": 8.0, "charged_usd": 0.009498265000000002 }, "presets": { "JevBench Score (25:25:25:25)": 73.02927314568292, "Balanced 33:33:33 (no calibration)": 75.75964805439804, "Emphasis on Accuracy 60:20:20": 78.45085411513585, "Emphasis on Speed 20:60:20": 81.65054154175054, "Emphasis on Cost 20:20:60": 67.8823861095801, "Intelligence only": 82.6679689875465 }, "rank": 3, "rank_under": { "JevBench Score (25:25:25:25)": 3, "Balanced 33:33:33 (no calibration)": 1, "Emphasis on Accuracy 60:20:20": 1, "Emphasis on Speed 20:60:20": 1, "Emphasis on Cost 20:20:60": 3, "Intelligence only": 9 } }, { "key": "winnow-12b", "display": "Winnow-12B Q8", "class": "jev-rebuild", "open": "yes", "author": "Eldan Ring", "repo": "https://huggingface.co/EldanRing/Winnow-12B", "licence": "Apache-2.0, including the applicable Gemma 4 base/derivative licence terms", "underlying": "google/gemma-4-12B-it LoRA fine-tune, merged and exported as Q8_0 GGUF", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our GPU (lium.io RTX 4090 24 GB), reached over the internet from Germany; serial, one request at a time", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.96875, "judge": 0.910958904109589, "hard": 0.7090909090909091 }, "axes": { "intelligence": 82.0447452273211, "calibration": 71.96875781135537, "speed": 82.32624042236164, "cost": 52.92322256021786 }, "jevbench_score": 71.21883011371422, "speed": { "p50_s_raw": 0.22503555566072464, "p95_s_raw": 0.4126893173903227, "p50_s_adjusted": 0.6000711113214493, "p95_s_adjusted": 0.9753786347806455, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.2846274711191654, "hard_tier_p95_s": 0.8760518249124287 }, "cost": { "kind": "estimate", "usd_per_1000": 0.037087359550561805, "basis": "ESTIMATE: hosted-provider price, OpenRouter google/gemma-3-12b-it hosted reference list price $0.05/M in, $0.0/M out (the nearest publicly hosted 12B Gemma sibling; Winnow reads answer logits in one forward pass and generates no answer tokens) x 393 input and 0 output tokens per decision (input tokens measured (the system's own count))", "usd_per_1000_v11_tiers": 0.01962611464968153, "usd_per_1000_hard": 0.06200931818181819, "self_host_sensitivity": null }, "calibration": { "score": 71.96875781135537, "score_label_only_as_onehot": null, "ece_hard": 0.11992766665864553, "probability_fidelity": 67.92304895443986, "brier_hard": 0.4051575575819251, "brier_standard_judge_v11": 0.09509412904044282, "note": null }, "hard": { "run": "runs/winnow-12b--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.7090909090909091, "by_family": { "adversarial": { "correct": 11, "n": 12, "accuracy": 0.9166666666666666 }, "ambiguous": { "correct": 12, "n": 14, "accuracy": 0.8571428571428571 }, "judge_hard": { "correct": 29, "n": 33, "accuracy": 0.8787878787878788 }, "long_policy": { "correct": 25, "n": 38, "accuracy": 0.6578947368421053 }, "multi_hop": { "correct": 24, "n": 35, "accuracy": 0.6857142857142857 }, "probability": { "correct": 11, "n": 20, "accuracy": 0.55 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 8, "n": 30, "accuracy": 0.26666666666666666 }, "tradeoff": { "correct": 10, "n": 12, "accuracy": 0.8333333333333334 }, "trap": { "correct": 16, "n": 16, "accuracy": 1.0 } }, "has_distribution": true, "brier_mean": 0.4051575575819251, "ece": 0.11992766665864553, "probability_fidelity": 67.92304895443986, "calibration_score": 71.96875781135537, "onehot": { "ece": 0.2909090909090909, "probability_fidelity": 47.842000000000006, "calibration_score": 44.83009090909091 }, "latency_p50_s": 0.2846274711191654, "latency_p95_s": 0.8760518249124287, "mean_input_tokens": 1240.1863636363637, "mean_output_tokens": 0.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 71.21883011371422, "Balanced 33:33:33 (no calibration)": 70.97059478306711, "Emphasis on Accuracy 60:20:20": 75.20857763442709, "Emphasis on Speed 20:60:20": 75.31168772062911, "Emphasis on Cost 20:20:60": 63.111075233063815, "Intelligence only": 82.0447452273211 }, "rank": 4, "rank_under": { "JevBench Score (25:25:25:25)": 4, "Balanced 33:33:33 (no calibration)": 4, "Emphasis on Accuracy 60:20:20": 4, "Emphasis on Speed 20:60:20": 5, "Emphasis on Cost 20:20:60": 10, "Intelligence only": 11 } }, { "key": "reflex-4b", "display": "reflex 4B (kshetrajna12)", "class": "jev-rebuild", "open": "yes", "author": "kshetrajna12", "repo": "https://github.com/kshetrajna12/reflex", "licence": "MIT (code, adapter); Apache-2.0 (base)", "underlying": "Qwen/Qwen3.5-4B + kshetrajna12/reflex-qwen3.5-4b-lora", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our RunPod GPU (H100 NVL 96 GB, Canada), reached over the internet from Germany", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.9479166666666666, "judge": 0.9726027397260274, "hard": 0.6318181818181818 }, "axes": { "intelligence": 80.13624687620394, "calibration": 75.21406363636362, "speed": 67.96813879081057, "cost": 59.675719111465405 }, "jevbench_score": 70.31658532999862, "speed": { "p50_s_raw": 1.7996779419481754, "p95_s_raw": 2.0541166696697473, "p50_s_adjusted": 3.7493558838963508, "p95_s_adjusted": 4.258233339339495, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 1.8708765655755997, "hard_tier_p95_s": 2.159583227336406 }, "cost": { "kind": "estimate", "usd_per_1000": 0.022087303370786515, "basis": "ESTIMATE: hosted-provider price, DeepInfra Qwen/Qwen3.5-4B list price $0.03/M in, $0.0/M out (the exact base weights; one pass, no generated output) x 377 input and 0 output tokens per decision (input tokens measured (the system's own count))", "usd_per_1000_v11_tiers": 0.01132184713375796, "usd_per_1000_hard": 0.037452545454545454, "self_host_sensitivity": null }, "calibration": { "score": 75.21406363636362, "score_label_only_as_onehot": null, "ece_hard": 0.10498626363636375, "probability_fidelity": 71.42537999999999, "brier_hard": 0.511049058192209, "brier_standard_judge_v11": 0.08995296889612805, "note": null }, "hard": { "run": "runs/reflex-4b--hard (job jevbench-add-requests-20260919, run 4)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.6318181818181818, "by_family": { "adversarial": { "correct": 10, "n": 12, "accuracy": 0.8333333333333334 }, "ambiguous": { "correct": 7, "n": 14, "accuracy": 0.5 }, "judge_hard": { "correct": 24, "n": 33, "accuracy": 0.7272727272727273 }, "long_policy": { "correct": 20, "n": 38, "accuracy": 0.5263157894736842 }, "multi_hop": { "correct": 22, "n": 35, "accuracy": 0.6285714285714286 }, "probability": { "correct": 11, "n": 20, "accuracy": 0.55 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 10, "n": 30, "accuracy": 0.3333333333333333 }, "tradeoff": { "correct": 9, "n": 12, "accuracy": 0.75 }, "trap": { "correct": 16, "n": 16, "accuracy": 1.0 } }, "has_distribution": true, "brier_mean": 0.511049058192209, "ece": 0.10498626363636375, "probability_fidelity": 71.42537999999999, "calibration_score": 75.21406363636362, "onehot": { "ece": 0.36818181818181817, "probability_fidelity": 49.4625, "calibration_score": 37.91306818181818 }, "latency_p50_s": 1.8708765655755997, "latency_p95_s": 2.159583227336406, "mean_input_tokens": 1248.418181818182, "mean_output_tokens": 0.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 70.31658532999862, "Balanced 33:33:33 (no calibration)": 68.7560120648147, "Emphasis on Accuracy 60:20:20": 73.1001060400747, "Emphasis on Speed 20:60:20": 68.43977269821107, "Emphasis on Cost 20:20:60": 64.96889374687606, "Intelligence only": 80.13624687620391 }, "rank": 5, "rank_under": { "JevBench Score (25:25:25:25)": 5, "Balanced 33:33:33 (no calibration)": 6, "Emphasis on Accuracy 60:20:20": 5, "Emphasis on Speed 20:60:20": 17, "Emphasis on Cost 20:20:60": 5, "Intelligence only": 13 } }, { "key": "jqv", "display": "jqv (Qwen3-32B zero-shot)", "class": "jev-rebuild", "open": "yes", "author": "hjmurmur (Octalab)", "repo": "https://github.com/Octalab-Inc/jqv", "licence": "Apache-2.0 (Qwen3-32B weights); serving code public", "underlying": "Qwen/Qwen3-32B, bf16, read as a direct-logit classifier (no fine-tuning)", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our RunPod GPU (H100 NVL 96 GB, Canada), reached over the internet from Germany", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.9583333333333334, "judge": 0.9246575342465754, "hard": 0.6454545454545455 }, "axes": { "intelligence": 79.28281068482198, "calibration": 79.0059752597162, "speed": 74.62291018872223, "cost": 47.458062754792564 }, "jevbench_score": 68.62862396499413, "speed": { "p50_s_raw": 0.7473905384540558, "p95_s_raw": 0.9735059145838022, "p50_s_adjusted": 1.6447810769081115, "p95_s_adjusted": 2.0970118291676045, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.8068596571683884, "hard_tier_p95_s": 1.5256738737225528 }, "cost": { "kind": "estimate", "usd_per_1000": 0.05641543071161049, "basis": "ESTIMATE: hosted-provider price, OpenRouter qwen/qwen3-32b list price $0.08/M in, $0.0/M out (the exact base model this system reads logits from; nothing is generated) x 359 input and 0 output tokens per decision (input tokens measured (the system's own count))", "usd_per_1000_v11_tiers": 0.02870751592356688, "usd_per_1000_hard": 0.09596218181818182, "self_host_sensitivity": null }, "calibration": { "score": 79.0059752597162, "score_label_only_as_onehot": null, "ece_hard": 0.08756655067713424, "probability_fidelity": 75.52526065485925, "brier_hard": 0.47242119377157676, "brier_standard_judge_v11": 0.13845653894523616, "note": null }, "hard": { "run": "runs/jqv--hard-r4 (job jevbench-add-requests-20260919, run 4)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.6454545454545455, "by_family": { "adversarial": { "correct": 12, "n": 12, "accuracy": 1.0 }, "ambiguous": { "correct": 9, "n": 14, "accuracy": 0.6428571428571429 }, "judge_hard": { "correct": 26, "n": 33, "accuracy": 0.7878787878787878 }, "long_policy": { "correct": 19, "n": 38, "accuracy": 0.5 }, "multi_hop": { "correct": 24, "n": 35, "accuracy": 0.6857142857142857 }, "probability": { "correct": 12, "n": 20, "accuracy": 0.6 }, "routing_hard": { "correct": 9, "n": 10, "accuracy": 0.9 }, "temporal_numeric": { "correct": 7, "n": 30, "accuracy": 0.23333333333333334 }, "tradeoff": { "correct": 8, "n": 12, "accuracy": 0.6666666666666666 }, "trap": { "correct": 16, "n": 16, "accuracy": 1.0 } }, "has_distribution": true, "brier_mean": 0.47242119377157676, "ece": 0.08756655067713424, "probability_fidelity": 75.52526065485925, "calibration_score": 79.0059752597162, "onehot": { "ece": 0.3545454545454545, "probability_fidelity": 51.3725, "calibration_score": 40.23170454545455 }, "latency_p50_s": 0.8068596571683884, "latency_p95_s": 1.5256738737225528, "mean_input_tokens": 1199.5272727272727, "mean_output_tokens": 0.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 68.62862396499413, "Balanced 33:33:33 (no calibration)": 65.48176980337307, "Emphasis on Accuracy 60:20:20": 70.68770153799821, "Emphasis on Speed 20:60:20": 68.99555588321653, "Emphasis on Cost 20:20:60": 57.57000242729401, "Intelligence only": 79.28281068482198 }, "rank": 6, "rank_under": { "JevBench Score (25:25:25:25)": 6, "Balanced 33:33:33 (no calibration)": 14, "Emphasis on Accuracy 60:20:20": 9, "Emphasis on Speed 20:60:20": 14, "Emphasis on Cost 20:20:60": 14, "Intelligence only": 16 } }, { "key": "decision-machine-1", "display": "decision-machine-1 (milliseconds.ai)", "class": "decision-api", "open": "no", "author": "milliseconds.ai (Baptiste Laget)", "repo": "https://www.milliseconds.ai", "licence": "proprietary API, closed weights", "underlying": "decision-machine-1 (weights not published)", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "production API (milliseconds.ai, served from its nearest region), measured from Germany", "endpoint_kind": "api", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.7604166666666666, "judge": 0.8972602739726028, "hard": 0.4681818181818182 }, "axes": { "intelligence": 62.07457007916058, "calibration": 70.4452442163124, "speed": 92.92504013916033, "cost": 53.668063581045736 }, "jevbench_score": 68.33662413233554, "speed": { "p50_s_raw": 0.17223640158772469, "p95_s_raw": 0.29605407454073424, "p50_s_adjusted": 0.17223640158772469, "p95_s_adjusted": 0.29605407454073424, "adjustment": "none (production API)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.18790292367339134, "hard_tier_p95_s": 0.3872291069477796 }, "cost": { "kind": "measured", "usd_per_1000": 0.035026591760299625, "basis": "public tariff x measured tokens: $0.04 per million input tokens, output free (https://docs.milliseconds.ai/reference/pricing, read 2026-09-21) x 496 input tokens per easy/standard/judge decision as reported by the API; the run used the free test key, the price is the paid one", "usd_per_1000_v11_tiers": 0.019859745222929936, "usd_per_1000_hard": 0.05667381818181818, "self_host_sensitivity": null }, "calibration": { "score": 70.4452442163124, "score_label_only_as_onehot": null, "ece_hard": 0.15839024302206123, "probability_fidelity": 72.56853703703703, "brier_hard": 0.6458498833513563, "brier_standard_judge_v11": 0.2808058520000086, "note": null }, "hard": { "run": "runs/decision-machine-1--hard (job jevbench-add-requests-20260919, run 4)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.4681818181818182, "by_family": { "adversarial": { "correct": 8, "n": 12, "accuracy": 0.6666666666666666 }, "ambiguous": { "correct": 3, "n": 14, "accuracy": 0.21428571428571427 }, "judge_hard": { "correct": 21, "n": 33, "accuracy": 0.6363636363636364 }, "long_policy": { "correct": 13, "n": 38, "accuracy": 0.34210526315789475 }, "multi_hop": { "correct": 16, "n": 35, "accuracy": 0.45714285714285713 }, "probability": { "correct": 9, "n": 20, "accuracy": 0.45 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 7, "n": 30, "accuracy": 0.23333333333333334 }, "tradeoff": { "correct": 4, "n": 12, "accuracy": 0.3333333333333333 }, "trap": { "correct": 12, "n": 16, "accuracy": 0.75 } }, "has_distribution": true, "brier_mean": 0.6458498833513563, "ece": 0.15839024302206123, "probability_fidelity": 72.56853703703703, "calibration_score": 70.4452442163124, "onehot": { "ece": 0.5318181818181817, "probability_fidelity": 45.02249999999999, "calibration_score": 22.511249999999993 }, "latency_p50_s": 0.18790292367339134, "latency_p95_s": 0.3872291069477796, "mean_input_tokens": 1416.8454545454545, "mean_output_tokens": 0.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 68.33662413233554, "Balanced 33:33:33 (no calibration)": 67.64787120670496, "Emphasis on Accuracy 60:20:20": 65.36089174443691, "Emphasis on Speed 20:60:20": 76.80784559163453, "Emphasis on Cost 20:20:60": 61.66501634000721, "Intelligence only": 62.07457007916059 }, "rank": 7, "rank_under": { "JevBench Score (25:25:25:25)": 7, "Balanced 33:33:33 (no calibration)": 9, "Emphasis on Accuracy 60:20:20": 22, "Emphasis on Speed 20:60:20": 3, "Emphasis on Cost 20:20:60": 11, "Intelligence only": 29 } }, { "key": "decider-35b-a3b", "display": "decider-35b-a3b (Mapika)", "class": "jev-rebuild", "open": "yes", "author": "Mapika", "repo": "https://huggingface.co/Mapika/decider-35b-a3b", "licence": "Apache-2.0", "underlying": "Qwen3.5-35B-A3B-Base with a trained decision readout, 34.7B parameters / 3B active", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our RunPod GPU (H100 NVL 96 GB), reached over the internet", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.96875, "judge": 0.910958904109589, "hard": 0.6545454545454545 }, "axes": { "intelligence": 79.57871029182618, "calibration": 71.50368181818185, "speed": 80.78478445085703, "cost": 45.30548488186928 }, "jevbench_score": 67.55405352545425, "speed": { "p50_s_raw": 0.29191894084215164, "p95_s_raw": 0.49371074326336223, "p50_s_adjusted": 0.7338378816843033, "p95_s_adjusted": 1.1374214865267245, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.3118191212415695, "hard_tier_p95_s": 0.6214927081018686 }, "cost": { "kind": "estimate", "usd_per_1000": 0.0665503745318352, "basis": "ESTIMATE: hosted-provider price, OpenRouter Qwen3.6-35B-A3B list price list price $0.1/M in, $0.0/M out (the closest public hosted 35B-A3B direct-logit model; no output is generated) x 312 input and 0 output tokens per decision (input tokens measured (the system's own count))", "usd_per_1000_v11_tiers": 0.031202547770700636, "usd_per_1000_hard": 0.11700136363636363, "self_host_sensitivity": null }, "calibration": { "score": 71.50368181818185, "score_label_only_as_onehot": null, "ece_hard": 0.18867318181818157, "probability_fidelity": 80.742, "brier_hard": 0.4866369675454547, "brier_standard_judge_v11": 0.1055856861570248, "note": null }, "hard": { "run": "runs/decider-35b-a3b--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.6545454545454545, "by_family": { "adversarial": { "correct": 12, "n": 12, "accuracy": 1.0 }, "ambiguous": { "correct": 9, "n": 14, "accuracy": 0.6428571428571429 }, "judge_hard": { "correct": 25, "n": 33, "accuracy": 0.7575757575757576 }, "long_policy": { "correct": 20, "n": 38, "accuracy": 0.5263157894736842 }, "multi_hop": { "correct": 26, "n": 35, "accuracy": 0.7428571428571429 }, "probability": { "correct": 12, "n": 20, "accuracy": 0.6 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 6, "n": 30, "accuracy": 0.2 }, "tradeoff": { "correct": 9, "n": 12, "accuracy": 0.75 }, "trap": { "correct": 15, "n": 16, "accuracy": 0.9375 } }, "has_distribution": true, "brier_mean": 0.4866369675454547, "ece": 0.18867318181818157, "probability_fidelity": 80.742, "calibration_score": 71.50368181818185, "onehot": { "ece": 0.34545454545454546, "probability_fidelity": 52.62050000000001, "calibration_score": 41.76479545454546 }, "latency_p50_s": 0.3118191212415695, "latency_p95_s": 0.6214927081018686, "mean_input_tokens": 1170.0136363636364, "mean_output_tokens": 0.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 67.55405352545425, "Balanced 33:33:33 (no calibration)": 66.28660096671378, "Emphasis on Accuracy 60:20:20": 71.31390341464208, "Emphasis on Speed 20:60:20": 71.74427944399763, "Emphasis on Cost 20:20:60": 56.926667789324846, "Intelligence only": 79.57871029182617 }, "rank": 8, "rank_under": { "JevBench Score (25:25:25:25)": 8, "Balanced 33:33:33 (no calibration)": 13, "Emphasis on Accuracy 60:20:20": 8, "Emphasis on Speed 20:60:20": 10, "Emphasis on Cost 20:20:60": 18, "Intelligence only": 14 } }, { "key": "open-alternative-jev", "display": "open-alternative-jev (Qwen3.5-4B, IkerMoel)", "class": "jev-rebuild", "open": "yes", "author": "IkerMoel", "repo": "https://github.com/ikermoel/open-alternative-jev", "licence": "Apache-2.0 (code and weights)", "underlying": "Qwen/Qwen3.5-4B (frozen, BF16)", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our RunPod GPU (RTX PRO 4500 Blackwell 32 GB (EU-RO-1)), reached over the internet", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.84375, "judge": 0.7465753424657534, "hard": 0.5681818181818182 }, "axes": { "intelligence": 64.04897795132875, "calibration": 63.16501249259078, "speed": 83.47780372641819, "cost": 59.62620384142349 }, "jevbench_score": 66.9883439146374, "speed": { "p50_s_raw": 0.20686038956046104, "p95_s_raw": 0.32322231084108355, "p50_s_adjusted": 0.5637207791209221, "p95_s_adjusted": 0.7964446216821671, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": "RunPod RTX PRO 4500 Blackwell 32 GB (EU-RO-1)", "measured_where": "as open-alternative-jev", "hard_tier_p50_s": 0.2411614954471588, "hard_tier_p95_s": 0.6258101891726254 }, "cost": { "kind": "estimate", "usd_per_1000": 0.022171404494382024, "basis": "ESTIMATE: hosted-provider price, deepinfra Qwen/Qwen3.5-4B list price $0.03/M in, $0.15/M out (as open-alternative-jev) x 383 input and 1 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts) | ESTIMATE: deepinfra Qwen/Qwen3.5-4B $0.03/M in, $0.15/M out x 1235 in / 1 out tokens per hard decision", "usd_per_1000_v11_tiers": 0.011652229299363059, "usd_per_1000_hard": 0.03718513636363636, "self_host_sensitivity": { "usd_per_1000": 0.036831988551922504, "score": 60.84437082593016, "machine": "1x RTX PRO 4500 Blackwell 32 GB (EU-RO-1) (on-demand, RunPod secure)", "usd_per_h": 0.72, "concurrency": 4, "utilisation": 0.3, "p50_s_used": 0.22099193131153502, "decisions_per_hour": 19548.22501600768 } }, "calibration": { "score": 63.16501249259078, "score_label_only_as_onehot": null, "ece_hard": 0.19036693270964333, "probability_fidelity": 64.40341152711024, "brier_hard": 0.6106512084932694, "brier_standard_judge_v11": 0.2681967810959945, "note": null }, "hard": { "run": "runs-gpu/open-alternative-jev-yesfirst--hard", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.5681818181818182, "by_family": { "adversarial": { "correct": 8, "n": 12, "accuracy": 0.6666666666666666 }, "ambiguous": { "correct": 9, "n": 14, "accuracy": 0.6428571428571429 }, "judge_hard": { "correct": 23, "n": 33, "accuracy": 0.696969696969697 }, "long_policy": { "correct": 17, "n": 38, "accuracy": 0.4473684210526316 }, "multi_hop": { "correct": 19, "n": 35, "accuracy": 0.5428571428571428 }, "probability": { "correct": 7, "n": 20, "accuracy": 0.35 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 9, "n": 30, "accuracy": 0.3 }, "tradeoff": { "correct": 7, "n": 12, "accuracy": 0.5833333333333334 }, "trap": { "correct": 16, "n": 16, "accuracy": 1.0 } }, "has_distribution": true, "brier_mean": 0.6106512084932694, "ece": 0.19036693270964333, "probability_fidelity": 64.40341152711024, "calibration_score": 63.16501249259078, "onehot": { "ece": 0.43181818181818177, "probability_fidelity": 42.308499999999995, "calibration_score": 27.97243181818182 }, "latency_p50_s": 0.2411614954471588, "latency_p95_s": 0.6258101891726254, "mean_input_tokens": null, "mean_output_tokens": null, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 66.9883439146374, "Balanced 33:33:33 (no calibration)": 68.31354030396389, "Emphasis on Accuracy 60:20:20": 66.57466003050608, "Emphasis on Speed 20:60:20": 74.01717092453922, "Emphasis on Cost 20:20:60": 64.69622491913037, "Intelligence only": 64.04897795132875 }, "run_key": "open-alternative-jev-yesfirst", "rank": 9, "rank_under": { "JevBench Score (25:25:25:25)": 9, "Balanced 33:33:33 (no calibration)": 7, "Emphasis on Accuracy 60:20:20": 17, "Emphasis on Speed 20:60:20": 6, "Emphasis on Cost 20:20:60": 8, "Intelligence only": 26 } }, { "key": "system-one-open", "display": "system-one-open (Gemma 4 E2B LoRA on an L4)", "class": "jev-rebuild", "open": "yes", "author": "mithalouni", "repo": "https://github.com/mithalouni/system-one-open", "licence": "MIT (repository LICENSE; Gemma weights keep Google’s terms)", "underlying": "google/gemma-4-E2B-it + LoRA", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "author's public demo endpoint (Modal, L4) — not a production service", "endpoint_kind": "demo", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.9375, "judge": 0.8767123287671232, "hard": 0.4909090909090909 }, "axes": { "intelligence": 69.54527897078596, "calibration": 56.699696211408835, "speed": 76.96012500732209, "cost": 64.82109224519658 }, "jevbench_score": 66.59745356122706, "speed": { "p50_s_raw": 0.6517308317124844, "p95_s_raw": 0.7724301926791667, "p50_s_adjusted": 1.3034616634249687, "p95_s_adjusted": 1.5448603853583334, "adjustment": "x2 (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.6776973158121109, "hard_tier_p95_s": 1.1177304897457356 }, "cost": { "kind": "estimate", "usd_per_1000": 0.014880936329588014, "basis": "ESTIMATE: hosted-provider price, deepinfra google/gemma-4-E4B-it list price $0.02/M in, $0.1/M out (Gemma 4 E2B is not listed; the nearest larger sibling, Gemma 4 E4B, is listed only on DeepInfra) x 383 input and 2 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | ESTIMATE: deepinfra google/gemma-4-E4B-it $0.02/M in, $0.1/M out x 1235 in / 2 out tokens per hard decision", "usd_per_1000_v11_tiers": 0.00786815286624204, "usd_per_1000_hard": 0.024890090909090907, "self_host_sensitivity": { "usd_per_1000": 0.06593618349183544, "score": 54.521904855816615, "machine": "1x L4 24 GB (Gemma 4 E2B + LoRA)", "usd_per_h": 0.43, "concurrency": 4, "utilisation": 0.3, "p50_s_used": 0.6624286341505328, "decisions_per_hour": 6521.45722163681 } }, "calibration": { "score": 56.699696211408835, "score_label_only_as_onehot": null, "ece_hard": 0.25707820493262257, "probability_fidelity": 64.81503340934218, "brier_hard": 0.7469110891631627, "brier_standard_judge_v11": 0.13816078684373112, "note": null }, "hard": { "run": "runs/system-one-open--hard", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.4909090909090909, "by_family": { "adversarial": { "correct": 9, "n": 12, "accuracy": 0.75 }, "ambiguous": { "correct": 6, "n": 14, "accuracy": 0.42857142857142855 }, "judge_hard": { "correct": 25, "n": 33, "accuracy": 0.7575757575757576 }, "long_policy": { "correct": 10, "n": 38, "accuracy": 0.2631578947368421 }, "multi_hop": { "correct": 15, "n": 35, "accuracy": 0.42857142857142855 }, "probability": { "correct": 9, "n": 20, "accuracy": 0.45 }, "routing_hard": { "correct": 9, "n": 10, "accuracy": 0.9 }, "temporal_numeric": { "correct": 8, "n": 30, "accuracy": 0.26666666666666666 }, "tradeoff": { "correct": 3, "n": 12, "accuracy": 0.25 }, "trap": { "correct": 14, "n": 16, "accuracy": 0.875 } }, "has_distribution": true, "brier_mean": 0.7469110891631627, "ece": 0.25707820493262257, "probability_fidelity": 64.81503340934218, "calibration_score": 56.699696211408835, "onehot": { "ece": 0.509090909090909, "probability_fidelity": 44.011500000000005, "calibration_score": 22.005750000000003 }, "latency_p50_s": 0.6776973158121109, "latency_p95_s": 1.1177304897457356, "mean_input_tokens": null, "mean_output_tokens": null, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 66.59745356122706, "Balanced 33:33:33 (no calibration)": 70.26675874188773, "Emphasis on Accuracy 60:20:20": 69.97727297796747, "Emphasis on Speed 20:60:20": 72.87125655243679, "Emphasis on Cost 20:20:60": 68.0356424969818, "Intelligence only": 69.54527897078594 }, "rank": 10, "rank_under": { "JevBench Score (25:25:25:25)": 10, "Balanced 33:33:33 (no calibration)": 5, "Emphasis on Accuracy 60:20:20": 11, "Emphasis on Speed 20:60:20": 9, "Emphasis on Cost 20:20:60": 2, "Intelligence only": 23 } }, { "key": "openjev-razorback16", "display": "OpenJev (DiffusionGemma 26B-A4B NVFP4, razorback16)", "class": "jev-rebuild", "open": "yes", "author": "razorback16 / Codiv", "repo": "https://github.com/razorback16/openjev", "licence": "Apache-2.0 (repo and weights)", "underlying": "nvidia/diffusiongemma-26B-A4B-it-NVFP4", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our RunPod GPU (RTX PRO 4500 Blackwell 32 GB (EU-RO-1)), reached over the internet", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.9583333333333334, "judge": 0.910958904109589, "hard": 0.6545454545454545 }, "axes": { "intelligence": 79.1518810235335, "calibration": 64.76011611808524, "speed": 83.1778984786352, "cost": 45.49198106171689 }, "jevbench_score": 66.36329072785742, "speed": { "p50_s_raw": 0.24127069488167763, "p95_s_raw": 0.3052692499011755, "p50_s_adjusted": 0.6325413897633553, "p95_s_adjusted": 0.760538499802351, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": "RunPod RTX PRO 4500 Blackwell 32 GB (EU-RO-1)", "measured_where": "from a Hetzner server in Germany over the internet to the pod's public TCP port (plain HTTP, one connection per request); model loaded before timing", "hard_tier_p50_s": 0.2737487629055977, "hard_tier_p95_s": 0.5951106011867523 }, "cost": { "kind": "estimate", "usd_per_1000": 0.06560455056179774, "basis": "ESTIMATE: hosted-provider price, openrouter google/gemma-4-26b-a4b-it list price $0.09/M in, $0.3/M out (DiffusionGemma 26B-A4B is not listed; the same-size Gemma 4 26B-A4B MoE sibling is (size class moe_26B-A4B)) x 380 input and 1 output tokens per decision (input tokens measured) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | ESTIMATE: openrouter google/gemma-4-26b-a4b-it $0.09/M in, $0.3/M out x 1222 in / 0 out tokens per hard decision", "usd_per_1000_v11_tiers": 0.034483662420382165, "usd_per_1000_hard": 0.11002254545454546, "self_host_sensitivity": { "usd_per_1000": 0.042441862057452644, "score": 59.305139262330144, "machine": "1x RTX PRO 4500 Blackwell 32 GB (EU-RO-1) (on-demand, RunPod secure)", "usd_per_h": 0.72, "concurrency": 4, "utilisation": 0.3, "p50_s_used": 0.25465117234471585, "decisions_per_hour": 16964.38292517306 } }, "calibration": { "score": 64.76011611808524, "score_label_only_as_onehot": null, "ece_hard": 0.17789083573394485, "probability_fidelity": 65.09839938295943, "brier_hard": 0.48442036776440417, "brier_standard_judge_v11": 0.12110731767657959, "note": null }, "hard": { "run": "runs-gpu/openjev-razorback16--hard", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.6545454545454545, "by_family": { "adversarial": { "correct": 12, "n": 12, "accuracy": 1.0 }, "ambiguous": { "correct": 10, "n": 14, "accuracy": 0.7142857142857143 }, "judge_hard": { "correct": 29, "n": 33, "accuracy": 0.8787878787878788 }, "long_policy": { "correct": 15, "n": 38, "accuracy": 0.39473684210526316 }, "multi_hop": { "correct": 28, "n": 35, "accuracy": 0.8 }, "probability": { "correct": 11, "n": 20, "accuracy": 0.55 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 6, "n": 30, "accuracy": 0.2 }, "tradeoff": { "correct": 8, "n": 12, "accuracy": 0.6666666666666666 }, "trap": { "correct": 15, "n": 16, "accuracy": 0.9375 } }, "has_distribution": true, "brier_mean": 0.48442036776440417, "ece": 0.17789083573394485, "probability_fidelity": 65.09839938295943, "calibration_score": 64.76011611808524, "onehot": { "ece": 0.34545454545454546, "probability_fidelity": 48.83850000000001, "calibration_score": 39.87379545454546 }, "latency_p50_s": 0.2737487629055977, "latency_p95_s": 0.5951106011867523, "mean_input_tokens": 1222.4727272727273, "mean_output_tokens": 0.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 66.36329072785742, "Balanced 33:33:33 (no calibration)": 66.9064536908185, "Emphasis on Accuracy 60:20:20": 71.55917134154046, "Emphasis on Speed 20:60:20": 72.9934656738483, "Emphasis on Cost 20:20:60": 57.339611561690404, "Intelligence only": 79.15188102353352 }, "rank": 11, "rank_under": { "JevBench Score (25:25:25:25)": 11, "Balanced 33:33:33 (no calibration)": 11, "Emphasis on Accuracy 60:20:20": 7, "Emphasis on Speed 20:60:20": 8, "Emphasis on Cost 20:20:60": 15, "Intelligence only": 17 } }, { "key": "simplejev-qwen3.8-27b", "display": "SimpleJev Qwen3.8-27B", "class": "jev-rebuild", "open": "yes", "author": "Featherless AI", "repo": "https://github.com/featherless-ai/simple-jev", "licence": "Apache-2.0 (Qwen weights); repository licence not stated", "underlying": "Qwen3.8-27B through SimpleJev's direct-logit classifier", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "author's public demo endpoint (Featherless Classifier Demo) — not a production service", "endpoint_kind": "demo", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.96875, "judge": 0.9315068493150684, "hard": 0.75 }, "axes": { "intelligence": 84.70717465474874, "calibration": 81.14491951146204, "speed": 71.17954293639775, "cost": 39.48943618731995 }, "jevbench_score": 66.29860868418298, "speed": { "p50_s_raw": 1.0138170085847378, "p95_s_raw": 1.8794299442321063, "p50_s_adjusted": 2.0276340171694756, "p95_s_adjusted": 3.7588598884642126, "adjustment": "x2 (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 1.5036342144012451, "hard_tier_p95_s": 2.3205738529562945 }, "cost": { "kind": "estimate", "usd_per_1000": 0.10399651685393257, "basis": "ESTIMATE: hosted-provider price, OpenRouter Gemma 4 26B-A4B size-class reference list price $0.09/M in, $0.0/M out (a public 27B dense model served as a direct-logit classifier; no output is generated) x 809 input and 0 output tokens per decision (input tokens measured (the system's own count))", "usd_per_1000_v11_tiers": 0.07279222929936303, "usd_per_1000_hard": 0.14853354545454545, "self_host_sensitivity": null }, "calibration": { "score": 81.14491951146204, "score_label_only_as_onehot": null, "ece_hard": 0.06112991220300847, "probability_fidelity": 74.51582146352575, "brier_hard": 0.2989611751696843, "brier_standard_judge_v11": 0.08157045921166492, "note": null }, "hard": { "run": "runs/simplejev-qwen3.8-27b--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.75, "by_family": { "adversarial": { "correct": 12, "n": 12, "accuracy": 1.0 }, "ambiguous": { "correct": 12, "n": 14, "accuracy": 0.8571428571428571 }, "judge_hard": { "correct": 28, "n": 33, "accuracy": 0.8484848484848485 }, "long_policy": { "correct": 26, "n": 38, "accuracy": 0.6842105263157895 }, "multi_hop": { "correct": 29, "n": 35, "accuracy": 0.8285714285714286 }, "probability": { "correct": 14, "n": 20, "accuracy": 0.7 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 8, "n": 30, "accuracy": 0.26666666666666666 }, "tradeoff": { "correct": 10, "n": 12, "accuracy": 0.8333333333333334 }, "trap": { "correct": 16, "n": 16, "accuracy": 1.0 } }, "has_distribution": true, "brier_mean": 0.2989611751696843, "ece": 0.06112991220300847, "probability_fidelity": 74.51582146352575, "calibration_score": 81.14491951146204, "onehot": { "ece": 0.25, "probability_fidelity": 55.716, "calibration_score": 52.858000000000004 }, "latency_p50_s": 1.5036342144012451, "latency_p95_s": 2.3205738529562945, "mean_input_tokens": 1650.3727272727272, "mean_output_tokens": 1.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 66.29860868418298, "Balanced 33:33:33 (no calibration)": 61.98007653146178, "Emphasis on Accuracy 60:20:20": 70.22946232157913, "Emphasis on Speed 20:60:20": 65.50784954295487, "Emphasis on Cost 20:20:60": 51.753966234371674, "Intelligence only": 84.70717465474871 }, "rank": 12, "rank_under": { "JevBench Score (25:25:25:25)": 12, "Balanced 33:33:33 (no calibration)": 18, "Emphasis on Accuracy 60:20:20": 10, "Emphasis on Speed 20:60:20": 24, "Emphasis on Cost 20:20:60": 22, "Intelligence only": 7 } }, { "key": "zerank-2", "display": "ZeroEntropy zerank-2", "class": "reranker", "open": "yes", "author": "ZeroEntropy", "repo": "https://huggingface.co/zeroentropy/zerank-2-reranker", "licence": "Apache-2.0", "underlying": "Qwen3-Reranker-derived 4B cross-encoder", "has_distribution": true, "probability_source": [ "public_calibrated_reranker_softmax" ], "endpoint_condition": "our GPU (lium.io A6000 48 GB), serial, one option batch per decision", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.7916666666666666, "judge": 0.8835616438356164, "hard": 0.4727272727272727 }, "axes": { "intelligence": 63.0186253114589, "calibration": 76.48899637062043, "speed": 78.96019184080492, "cost": 49.75473753033334 }, "jevbench_score": 65.96713617823673, "speed": { "p50_s_raw": 0.12663885252550244, "p95_s_raw": 1.5002395501825958, "p50_s_adjusted": 0.4032777050510049, "p95_s_adjusted": 3.1504791003651915, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.23386710975319147, "hard_tier_p95_s": 1.8308645181823509 }, "cost": { "kind": "measured", "usd_per_1000": 0.04729792435014604, "basis": "MEASURED model inference time x lium.io A6000 tariff USD 0.42/hour", "usd_per_1000_v11_tiers": 0.03125511838232944, "usd_per_1000_hard": 0.07019538377693882, "self_host_sensitivity": "Whole five-model rental including setup/download was USD 0.21." }, "calibration": { "score": 76.48899637062043, "score_label_only_as_onehot": null, "ece_hard": 0.09786559872743003, "probability_fidelity": 72.55111248672686, "brier_hard": 0.631716160803407, "brier_standard_judge_v11": 0.40128849491331786, "note": null }, "hard": { "run": "runs/zerank-2--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.4727272727272727, "by_family": { "adversarial": { "correct": 7, "n": 12, "accuracy": 0.5833333333333334 }, "ambiguous": { "correct": 4, "n": 14, "accuracy": 0.2857142857142857 }, "judge_hard": { "correct": 18, "n": 33, "accuracy": 0.5454545454545454 }, "long_policy": { "correct": 12, "n": 38, "accuracy": 0.3157894736842105 }, "multi_hop": { "correct": 19, "n": 35, "accuracy": 0.5428571428571428 }, "probability": { "correct": 8, "n": 20, "accuracy": 0.4 }, "routing_hard": { "correct": 9, "n": 10, "accuracy": 0.9 }, "temporal_numeric": { "correct": 7, "n": 30, "accuracy": 0.23333333333333334 }, "tradeoff": { "correct": 8, "n": 12, "accuracy": 0.6666666666666666 }, "trap": { "correct": 12, "n": 16, "accuracy": 0.75 } }, "has_distribution": true, "brier_mean": 0.631716160803407, "ece": 0.09786559872743003, "probability_fidelity": 72.55111248672686, "calibration_score": 76.48899637062043, "onehot": { "ece": 0.5272727272727273, "probability_fidelity": 38.704499999999996, "calibration_score": 19.352249999999998 }, "latency_p50_s": 0.23386710975319147, "latency_p95_s": 1.8308645181823509, "mean_input_tokens": 4350.459090909091, "mean_output_tokens": 0.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 65.96713617823673, "Balanced 33:33:33 (no calibration)": 62.79193307669525, "Emphasis on Accuracy 60:20:20": 62.88251195039167, "Emphasis on Speed 20:60:20": 68.81856141109957, "Emphasis on Cost 20:20:60": 57.210545366210276, "Intelligence only": 63.018625311458905 }, "rank": 13, "rank_under": { "JevBench Score (25:25:25:25)": 13, "Balanced 33:33:33 (no calibration)": 15, "Emphasis on Accuracy 60:20:20": 29, "Emphasis on Speed 20:60:20": 15, "Emphasis on Cost 20:20:60": 16, "Intelligence only": 28 } }, { "key": "gpt-5.6-luna", "display": "GPT-5.6 Luna (low reasoning effort)", "class": "llm-baseline", "open": "no", "author": "OpenAI", "repo": null, "licence": "proprietary API", "underlying": "closed", "has_distribution": true, "probability_source": [ "verbalized" ], "endpoint_condition": "production API (OpenAI), reasoning effort low", "endpoint_kind": "api", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.9791666666666666, "judge": 0.9657534246575342, "hard": 0.9454545454545454 }, "axes": { "intelligence": 95.3254678182423, "calibration": 89.79207658196586, "speed": 77.5464346646158, "cost": 28.490318789992514 }, "jevbench_score": 65.94418818052425, "speed": { "p50_s_raw": 0.9680032916367054, "p95_s_raw": 1.8175220962613816, "p50_s_adjusted": 0.9680032916367054, "p95_s_adjusted": 1.8175220962613816, "adjustment": "none (production API)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 1.2202091813087463, "hard_tier_p95_s": 3.4094011016190024 }, "cost": { "kind": "measured", "usd_per_1000": 0.24191123595505612, "basis": "public tariff x measured tokens (https://platform.openai.com/docs/pricing (standard tier, read 2026-09-19)) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | public tariff x measured tokens (hard-tier run)", "usd_per_1000_v11_tiers": 0.15501464968152867, "usd_per_1000_hard": 0.3659363636363635, "self_host_sensitivity": null }, "calibration": { "score": 89.79207658196586, "score_label_only_as_onehot": null, "ece_hard": 0.07068160957409142, "probability_fidelity": 93.72047507875, "brier_hard": 0.11785378708136751, "brier_standard_judge_v11": 0.05621211608720952, "note": null }, "hard": { "run": "runs/gpt-5.6-luna--hard", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.9454545454545454, "by_family": { "adversarial": { "correct": 12, "n": 12, "accuracy": 1.0 }, "ambiguous": { "correct": 13, "n": 14, "accuracy": 0.9285714285714286 }, "judge_hard": { "correct": 30, "n": 33, "accuracy": 0.9090909090909091 }, "long_policy": { "correct": 36, "n": 38, "accuracy": 0.9473684210526315 }, "multi_hop": { "correct": 33, "n": 35, "accuracy": 0.9428571428571428 }, "probability": { "correct": 18, "n": 20, "accuracy": 0.9 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 28, "n": 30, "accuracy": 0.9333333333333333 }, "tradeoff": { "correct": 12, "n": 12, "accuracy": 1.0 }, "trap": { "correct": 16, "n": 16, "accuracy": 1.0 } }, "has_distribution": true, "brier_mean": 0.11785378708136751, "ece": 0.07068160957409142, "probability_fidelity": 93.72047507875, "calibration_score": 89.79207658196586, "onehot": { "ece": 0.054545454545454564, "probability_fidelity": 62.242, "calibration_score": 75.66645454545454 }, "latency_p50_s": 1.2202091813087463, "latency_p95_s": 3.4094011016190024, "mean_input_tokens": 1185.5545454545454, "mean_output_tokens": 107.35454545454546, "charged_usd": 0.08050599999999997 }, "presets": { "JevBench Score (25:25:25:25)": 65.94418818052425, "Balanced 33:33:33 (no calibration)": 59.49621845044213, "Emphasis on Accuracy 60:20:20": 71.84179849311023, "Emphasis on Speed 20:60:20": 66.14824890335301, "Emphasis on Cost 20:20:60": 44.31722325293679, "Intelligence only": 95.32546781824233 }, "rank": 14, "rank_under": { "JevBench Score (25:25:25:25)": 14, "Balanced 33:33:33 (no calibration)": 23, "Emphasis on Accuracy 60:20:20": 6, "Emphasis on Speed 20:60:20": 23, "Emphasis on Cost 20:20:60": 30, "Intelligence only": 1 } }, { "key": "openjev-sglang", "display": "openjev-sglang (Qwen3.6-35B-A3B on SGLang)", "class": "jev-rebuild", "open": "yes", "author": "ekzhang", "repo": "https://github.com/ekzhang/openjev-sglang", "licence": "no licence file in the repository as of 2026-09-19; Qwen3.6 weights keep their own terms", "underlying": "Qwen3.6-35B-A3B", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "author's public demo endpoint (Modal) — not a production service", "endpoint_kind": "demo", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.9583333333333334, "judge": 0.952054794520548, "hard": 0.7136363636363636 }, "axes": { "intelligence": 83.44922532193256, "calibration": 77.41176873901938, "speed": 77.059799513554, "cost": 36.45449272670061 }, "jevbench_score": 65.26826354532687, "speed": { "p50_s_raw": 0.6776718497276306, "p95_s_raw": 0.7260066717863083, "p50_s_adjusted": 1.3553436994552612, "p95_s_adjusted": 1.4520133435726166, "adjustment": "x2 (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.6935733407735825, "hard_tier_p95_s": 0.8754492454230784 }, "cost": { "kind": "estimate", "usd_per_1000": 0.13127546816479402, "basis": "ESTIMATE: hosted-provider price, openrouter qwen/qwen3.6-35b-a3b list price $0.1/M in, $0.9/M out (same base weights) x 610 input and 2 output tokens per decision [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | ESTIMATE: openrouter qwen/qwen3.6-35b-a3b $0.1/M in, $0.9/M out x 2272 in / 2 out tokens per hard decision", "usd_per_1000_v11_tiers": 0.0628296178343949, "usd_per_1000_hard": 0.22896636363636363, "self_host_sensitivity": { "usd_per_1000": 0.13621106532407892, "score": 46.64469025955533, "machine": "1x L40S 48 GB (Qwen3.6-35B-A3B, SGLang)", "usd_per_h": 0.86, "concurrency": 4, "utilisation": 0.3, "p50_s_used": 0.6842230258139779, "decisions_per_hour": 6313.7308114715415 } }, "calibration": { "score": 77.41176873901938, "score_label_only_as_onehot": null, "ece_hard": 0.09423841992624496, "probability_fidelity": 73.67122146328776, "brier_hard": 0.4005126115129668, "brier_standard_judge_v11": 0.08533204520463213, "note": null }, "hard": { "run": "runs/openjev-sglang--hard", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.7136363636363636, "by_family": { "adversarial": { "correct": 12, "n": 12, "accuracy": 1.0 }, "ambiguous": { "correct": 11, "n": 14, "accuracy": 0.7857142857142857 }, "judge_hard": { "correct": 27, "n": 33, "accuracy": 0.8181818181818182 }, "long_policy": { "correct": 23, "n": 38, "accuracy": 0.6052631578947368 }, "multi_hop": { "correct": 28, "n": 35, "accuracy": 0.8 }, "probability": { "correct": 11, "n": 20, "accuracy": 0.55 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 12, "n": 30, "accuracy": 0.4 }, "tradeoff": { "correct": 8, "n": 12, "accuracy": 0.6666666666666666 }, "trap": { "correct": 15, "n": 16, "accuracy": 0.9375 } }, "has_distribution": true, "brier_mean": 0.4005126115129668, "ece": 0.09423841992624496, "probability_fidelity": 73.67122146328776, "calibration_score": 77.41176873901938, "onehot": { "ece": 0.2863636363636364, "probability_fidelity": 49.77550000000001, "calibration_score": 46.25138636363637 }, "latency_p50_s": 0.6935733407735825, "latency_p95_s": 0.8754492454230784, "mean_input_tokens": 2271.663636363636, "mean_output_tokens": 2.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 65.26826354532687, "Balanced 33:33:33 (no calibration)": 61.65955903309022, "Emphasis on Accuracy 60:20:20": 69.59357913494426, "Emphasis on Speed 20:60:20": 67.41109762184071, "Emphasis on Cost 20:20:60": 49.96900162904047, "Intelligence only": 83.44922532193253 }, "rank": 15, "rank_under": { "JevBench Score (25:25:25:25)": 15, "Balanced 33:33:33 (no calibration)": 19, "Emphasis on Accuracy 60:20:20": 12, "Emphasis on Speed 20:60:20": 18, "Emphasis on Cost 20:20:60": 24, "Intelligence only": 8 } }, { "key": "qwen3-reranker-4b", "display": "Qwen3-Reranker-4B", "class": "reranker", "open": "yes", "author": "Qwen", "repo": "https://huggingface.co/Qwen/Qwen3-Reranker-4B", "licence": "Apache-2.0", "underlying": "4B instruction-aware generative reranker", "has_distribution": true, "probability_source": [ "public_calibrated_reranker_softmax" ], "endpoint_condition": "our GPU (lium.io A6000 48 GB), serial, one option batch per decision", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.7916666666666666, "judge": 0.8767123287671232, "hard": 0.5 }, "axes": { "intelligence": 63.98067503727088, "calibration": 67.03907338325979, "speed": 78.72320187600475, "cost": 49.152310354989595 }, "jevbench_score": 63.82721228923758, "speed": { "p50_s_raw": 0.1302479188889265, "p95_s_raw": 1.5593349148519338, "p50_s_adjusted": 0.41049583777785303, "p95_s_adjusted": 3.2686698297038674, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.2625806527212262, "hard_tier_p95_s": 1.8920937991235407 }, "cost": { "kind": "measured", "usd_per_1000": 0.04953623422454652, "basis": "MEASURED model inference time x lium.io A6000 tariff USD 0.42/hour", "usd_per_1000_v11_tiers": 0.03307339728309628, "usd_per_1000_hard": 0.07303319240461638, "self_host_sensitivity": "Whole five-model rental including setup/download was USD 0.21." }, "calibration": { "score": 67.03907338325979, "score_label_only_as_onehot": null, "ece_hard": 0.19149945567314672, "probability_fidelity": 72.37803790114894, "brier_hard": 0.6573086779280183, "brier_standard_judge_v11": 0.23321289738391712, "note": null }, "hard": { "run": "runs/qwen3-reranker-4b--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.5, "by_family": { "adversarial": { "correct": 8, "n": 12, "accuracy": 0.6666666666666666 }, "ambiguous": { "correct": 6, "n": 14, "accuracy": 0.42857142857142855 }, "judge_hard": { "correct": 18, "n": 33, "accuracy": 0.5454545454545454 }, "long_policy": { "correct": 16, "n": 38, "accuracy": 0.42105263157894735 }, "multi_hop": { "correct": 18, "n": 35, "accuracy": 0.5142857142857142 }, "probability": { "correct": 11, "n": 20, "accuracy": 0.55 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 8, "n": 30, "accuracy": 0.26666666666666666 }, "tradeoff": { "correct": 5, "n": 12, "accuracy": 0.4166666666666667 }, "trap": { "correct": 10, "n": 16, "accuracy": 0.625 } }, "has_distribution": true, "brier_mean": 0.6573086779280183, "ece": 0.19149945567314672, "probability_fidelity": 72.37803790114894, "calibration_score": 67.03907338325979, "onehot": { "ece": 0.5, "probability_fidelity": 49.763999999999996, "calibration_score": 24.881999999999998 }, "latency_p50_s": 0.2625806527212262, "latency_p95_s": 1.8920937991235407, "mean_input_tokens": 4350.459090909091, "mean_output_tokens": 0.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 63.82721228923758, "Balanced 33:33:33 (no calibration)": 62.79115927506405, "Emphasis on Accuracy 60:20:20": 63.26428846637759, "Emphasis on Speed 20:60:20": 68.73535825821553, "Emphasis on Cost 20:20:60": 56.932030748003164, "Intelligence only": 63.980675037270856 }, "rank": 16, "rank_under": { "JevBench Score (25:25:25:25)": 16, "Balanced 33:33:33 (no calibration)": 16, "Emphasis on Accuracy 60:20:20": 27, "Emphasis on Speed 20:60:20": 16, "Emphasis on Cost 20:20:60": 17, "Intelligence only": 27 } }, { "key": "reflex-27b", "display": "reflex-27b (Qwen3.8-27B)", "class": "jev-rebuild", "open": "yes", "author": "kshetrajna12", "repo": "https://github.com/kshetrajna12/reflex", "licence": "MIT code; Apache-2.0 Qwen weights", "underlying": "Qwen3.8-27B, frozen, direct-logit readout averaged across two option orders", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our RunPod GPU (H100 NVL 96 GB), reached over the internet", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.9583333333333334, "judge": 0.958904109589041, "hard": 0.759090909090909 }, "axes": { "intelligence": 85.77522217678047, "calibration": 86.17813284090909, "speed": 67.46599369477367, "cost": 32.26119927292868 }, "jevbench_score": 63.33315017675085, "speed": { "p50_s_raw": 1.8879061974585056, "p95_s_raw": 2.2076592873781915, "p50_s_adjusted": 3.925812394917011, "p95_s_adjusted": 4.565318574756383, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 2.1106490306556225, "hard_tier_p95_s": 2.5633741833269594 }, "cost": { "kind": "estimate", "usd_per_1000": 0.18111733707865169, "basis": "ESTIMATE: hosted-provider price, OpenRouter Qwen3.8-27B list price list price $0.214/M in, $0.0/M out (the exact public base weights used as a direct-logit classifier; no output is generated) x 481 input and 0 output tokens per decision (input tokens measured (the system's own count))", "usd_per_1000_v11_tiers": 0.10295512738853504, "usd_per_1000_hard": 0.29267612727272724, "self_host_sensitivity": null }, "calibration": { "score": 86.17813284090909, "score_label_only_as_onehot": null, "ece_hard": 0.046629459090909105, "probability_fidelity": 81.6821575, "brier_hard": 0.3173506562125362, "brier_standard_judge_v11": 0.0736030978986818, "note": null }, "hard": { "run": "runs/reflex-27b--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.759090909090909, "by_family": { "adversarial": { "correct": 12, "n": 12, "accuracy": 1.0 }, "ambiguous": { "correct": 13, "n": 14, "accuracy": 0.9285714285714286 }, "judge_hard": { "correct": 27, "n": 33, "accuracy": 0.8181818181818182 }, "long_policy": { "correct": 26, "n": 38, "accuracy": 0.6842105263157895 }, "multi_hop": { "correct": 28, "n": 35, "accuracy": 0.8 }, "probability": { "correct": 13, "n": 20, "accuracy": 0.65 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 12, "n": 30, "accuracy": 0.4 }, "tradeoff": { "correct": 10, "n": 12, "accuracy": 0.8333333333333334 }, "trap": { "correct": 16, "n": 16, "accuracy": 1.0 } }, "has_distribution": true, "brier_mean": 0.3173506562125362, "ece": 0.046629459090909105, "probability_fidelity": 81.6821575, "calibration_score": 86.17813284090909, "onehot": { "ece": 0.24090909090909096, "probability_fidelity": 53.07349999999998, "calibration_score": 52.44584090909089 }, "latency_p50_s": 2.1106490306556225, "latency_p95_s": 2.5633741833269594, "mean_input_tokens": 1367.6454545454546, "mean_output_tokens": 0.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 63.33315017675085, "Balanced 33:33:33 (no calibration)": 57.15344680707083, "Emphasis on Accuracy 60:20:20": 67.23109687384421, "Emphasis on Speed 20:60:20": 61.074430318668846, "Emphasis on Cost 20:20:60": 45.46714234858062, "Intelligence only": 85.77522217678043 }, "rank": 17, "rank_under": { "JevBench Score (25:25:25:25)": 17, "Balanced 33:33:33 (no calibration)": 26, "Emphasis on Accuracy 60:20:20": 16, "Emphasis on Speed 20:60:20": 28, "Emphasis on Cost 20:20:60": 27, "Intelligence only": 4 } }, { "key": "litjev", "display": "LitJev (Qwen3.8-27B)", "class": "jev-rebuild", "open": "yes", "author": "Zhengxu Yu", "repo": "https://github.com/zhengxuyu/litjev", "licence": "Apache-2.0 (code); Apache-2.0 base weights", "underlying": "Qwen/Qwen3.8-27B, frozen, read at the output head", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our RunPod GPU (H100 NVL 96 GB, Canada), reached over the internet from Germany", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.9791666666666666, "judge": 0.8835616438356164, "hard": 0.7318181818181818 }, "axes": { "intelligence": 82.41521808432806, "calibration": 83.51703479356011, "speed": 66.72450940291681, "cost": 33.63149944045428 }, "jevbench_score": 62.69075853252832, "speed": { "p50_s_raw": 2.025244139134884, "p95_s_raw": 2.4555754292756315, "p50_s_adjusted": 4.200488278269768, "p95_s_adjusted": 5.0611508585512635, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 2.2890688702464104, "hard_tier_p95_s": 2.9710183084011077 }, "cost": { "kind": "estimate", "usd_per_1000": 0.16303594007490638, "basis": "ESTIMATE: hosted-provider price, OpenRouter Qwen3.8-27B (as the reflex-27b row) list price $0.214/M in, $0.0/M out (the exact base weights; nothing is generated) x 418 input and 0 output tokens per decision (input tokens measured (the system's own count))", "usd_per_1000_v11_tiers": 0.08940292993630572, "usd_per_1000_hard": 0.2681303272727273, "self_host_sensitivity": null }, "calibration": { "score": 83.51703479356011, "score_label_only_as_onehot": null, "ece_hard": 0.047705096932661215, "probability_fidelity": 76.57508897365247, "brier_hard": 0.3296182961504934, "brier_standard_judge_v11": 0.11729020229951063, "note": null }, "hard": { "run": "runs/litjev--hard (job jevbench-add-requests-20260919, run 4)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.7318181818181818, "by_family": { "adversarial": { "correct": 12, "n": 12, "accuracy": 1.0 }, "ambiguous": { "correct": 12, "n": 14, "accuracy": 0.8571428571428571 }, "judge_hard": { "correct": 24, "n": 33, "accuracy": 0.7272727272727273 }, "long_policy": { "correct": 26, "n": 38, "accuracy": 0.6842105263157895 }, "multi_hop": { "correct": 29, "n": 35, "accuracy": 0.8285714285714286 }, "probability": { "correct": 13, "n": 20, "accuracy": 0.65 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 9, "n": 30, "accuracy": 0.3 }, "tradeoff": { "correct": 10, "n": 12, "accuracy": 0.8333333333333334 }, "trap": { "correct": 16, "n": 16, "accuracy": 1.0 } }, "has_distribution": true, "brier_mean": 0.3296182961504934, "ece": 0.047705096932661215, "probability_fidelity": 76.57508897365247, "calibration_score": 83.51703479356011, "onehot": { "ece": 0.2681818181818182, "probability_fidelity": 53.50899999999999, "calibration_score": 49.93631818181818 }, "latency_p50_s": 2.2890688702464104, "latency_p95_s": 2.9710183084011077, "mean_input_tokens": 1252.9454545454546, "mean_output_tokens": 0.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 62.69075853252832, "Balanced 33:33:33 (no calibration)": 56.974389114496205, "Emphasis on Accuracy 60:20:20": 66.04056364973486, "Emphasis on Speed 20:60:20": 60.6906742148906, "Emphasis on Cost 20:20:60": 46.1430501192921, "Intelligence only": 82.41521808432807 }, "rank": 18, "rank_under": { "JevBench Score (25:25:25:25)": 18, "Balanced 33:33:33 (no calibration)": 28, "Emphasis on Accuracy 60:20:20": 19, "Emphasis on Speed 20:60:20": 29, "Emphasis on Cost 20:20:60": 26, "Intelligence only": 10 } }, { "key": "kev-0.6b", "display": "kev 0.6B (research preview)", "class": "jev-rebuild", "open": true, "author": "Jared Palmer", "repo": "https://github.com/jaredpalmer/kev", "licence": "Apache-2.0", "underlying": "Qwen3-0.6B-Base + LoRA + learned pointer head; jaredpalmer/kev-0.6b", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our RunPod GPU (GeForce RTX 3090 24 GB, community cloud CA), reached over the internet", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.8125, "judge": 0.6643835616438356, "hard": 0.4 }, "axes": { "intelligence": 51.9132695254489, "calibration": 51.092162895835045, "speed": 75.55649753874303, "cost": 76.08691668696139 }, "jevbench_score": 62.489006255662105, "speed": { "p50_s_raw": 0.5904003903269768, "p95_s_raw": 0.9702187780290842, "p50_s_adjusted": 1.3308007806539535, "p95_s_adjusted": 2.0904375560581685, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.6060735657811165, "hard_tier_p95_s": 1.4071057934314009 }, "cost": { "kind": "estimate", "usd_per_1000": 0.0062676217228464426, "basis": "ESTIMATE: hosted-provider price, DeepInfra Qwen3-Embedding-0.6B size-class reference list price $0.01/M in, $0.0/M out (a <=0.6B one-pass model with no generated output) x 279 input and 0 output tokens per decision (input tokens measured (the system's own count))", "usd_per_1000_v11_tiers": 0.0027941082802547773, "usd_per_1000_hard": 0.011225272727272728, "self_host_sensitivity": null }, "calibration": { "score": 51.092162895835045, "score_label_only_as_onehot": null, "ece_hard": 0.26937623307785324, "probability_fidelity": 56.05957240724073, "brier_hard": 0.8234993914874135, "brier_standard_judge_v11": 0.3857487253984507, "note": null }, "hard": { "run": "runs/kev-0.6b--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.4, "by_family": { "adversarial": { "correct": 8, "n": 12, "accuracy": 0.6666666666666666 }, "ambiguous": { "correct": 4, "n": 14, "accuracy": 0.2857142857142857 }, "judge_hard": { "correct": 18, "n": 33, "accuracy": 0.5454545454545454 }, "long_policy": { "correct": 11, "n": 38, "accuracy": 0.2894736842105263 }, "multi_hop": { "correct": 11, "n": 35, "accuracy": 0.3142857142857143 }, "probability": { "correct": 6, "n": 20, "accuracy": 0.3 }, "routing_hard": { "correct": 5, "n": 10, "accuracy": 0.5 }, "temporal_numeric": { "correct": 10, "n": 30, "accuracy": 0.3333333333333333 }, "tradeoff": { "correct": 6, "n": 12, "accuracy": 0.5 }, "trap": { "correct": 9, "n": 16, "accuracy": 0.5625 } }, "has_distribution": true, "brier_mean": 0.8234993914874135, "ece": 0.26937623307785324, "probability_fidelity": 56.05957240724073, "calibration_score": 51.092162895835045, "onehot": { "ece": 0.6, "probability_fidelity": 34.61249999999999, "calibration_score": 17.306249999999995 }, "latency_p50_s": 0.6060735657811165, "latency_p95_s": 1.4071057934314009, "mean_input_tokens": 1122.5272727272727, "mean_output_tokens": 62.20909090909091, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 62.489006255662105, "Balanced 33:33:33 (no calibration)": 66.82722000587596, "Emphasis on Accuracy 60:20:20": 60.4064605844487, "Emphasis on Speed 20:60:20": 70.19089241021678, "Emphasis on Cost 20:20:60": 70.38757953750113, "Intelligence only": 51.91326952544889 }, "rank": 19, "rank_under": { "JevBench Score (25:25:25:25)": 19, "Balanced 33:33:33 (no calibration)": 12, "Emphasis on Accuracy 60:20:20": 30, "Emphasis on Speed 20:60:20": 13, "Emphasis on Cost 20:20:60": 1, "Intelligence only": 32 } }, { "key": "simplejev-qwen3.6-35b-a3b", "display": "SimpleJev Qwen3.6-35B-A3B", "class": "jev-rebuild", "open": "yes", "author": "Featherless AI", "repo": "https://github.com/featherless-ai/simple-jev", "licence": "Apache-2.0 (Qwen weights); repository licence not stated", "underlying": "Qwen3.6-35B-A3B through SimpleJev's direct-logit classifier", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "author's public demo endpoint (Featherless Classifier Demo) — not a production service", "endpoint_kind": "demo", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.9375, "judge": 0.9315068493150684, "hard": 0.6636363636363637 }, "axes": { "intelligence": 79.52213153533707, "calibration": 67.05838957604325, "speed": 74.98811602029707, "cost": 38.11671147049971 }, "jevbench_score": 62.48305419264297, "speed": { "p50_s_raw": 0.8519573211669922, "p95_s_raw": 0.9304875515401363, "p50_s_adjusted": 1.7039146423339844, "p95_s_adjusted": 1.8609751030802726, "adjustment": "x2 (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.8764849826693535, "hard_tier_p95_s": 0.9821765016764402 }, "cost": { "kind": "estimate", "usd_per_1000": 0.11555168539325843, "basis": "ESTIMATE: hosted-provider price, OpenRouter Qwen3.6-35B-A3B list price list price $0.1/M in, $0.0/M out (the same base weights served as a direct-logit classifier; no output is generated) x 809 input and 0 output tokens per decision (input tokens measured (the system's own count))", "usd_per_1000_v11_tiers": 0.08088025477707006, "usd_per_1000_hard": 0.16503727272727273, "self_host_sensitivity": null }, "calibration": { "score": 67.05838957604325, "score_label_only_as_onehot": null, "ece_hard": 0.16640343503383082, "probability_fidelity": 67.39746615885267, "brier_hard": 0.4791052568647208, "brier_standard_judge_v11": 0.0940923781582149, "note": null }, "hard": { "run": "runs/simplejev-qwen3.6-35b-a3b--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.6636363636363637, "by_family": { "adversarial": { "correct": 12, "n": 12, "accuracy": 1.0 }, "ambiguous": { "correct": 8, "n": 14, "accuracy": 0.5714285714285714 }, "judge_hard": { "correct": 24, "n": 33, "accuracy": 0.7272727272727273 }, "long_policy": { "correct": 23, "n": 38, "accuracy": 0.6052631578947368 }, "multi_hop": { "correct": 26, "n": 35, "accuracy": 0.7428571428571429 }, "probability": { "correct": 10, "n": 20, "accuracy": 0.5 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 10, "n": 30, "accuracy": 0.3333333333333333 }, "tradeoff": { "correct": 8, "n": 12, "accuracy": 0.6666666666666666 }, "trap": { "correct": 15, "n": 16, "accuracy": 0.9375 } }, "has_distribution": true, "brier_mean": 0.4791052568647208, "ece": 0.16640343503383082, "probability_fidelity": 67.39746615885267, "calibration_score": 67.05838957604325, "onehot": { "ece": 0.3363636363636363, "probability_fidelity": 47.1915, "calibration_score": 39.95938636363637 }, "latency_p50_s": 0.8764849826693535, "latency_p95_s": 0.9821765016764402, "mean_input_tokens": 1650.3727272727272, "mean_output_tokens": 1.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 62.48305419264297, "Balanced 33:33:33 (no calibration)": 61.02839661038186, "Emphasis on Accuracy 60:20:20": 67.84445857465217, "Emphasis on Speed 20:60:20": 66.26987524691448, "Emphasis on Cost 20:20:60": 50.55514331281906, "Intelligence only": 79.52213153533707 }, "rank": 20, "rank_under": { "JevBench Score (25:25:25:25)": 20, "Balanced 33:33:33 (no calibration)": 21, "Emphasis on Accuracy 60:20:20": 14, "Emphasis on Speed 20:60:20": 21, "Emphasis on Cost 20:20:60": 23, "Intelligence only": 15 } }, { "key": "djev-thinking", "display": "djev (thinking)", "class": "jev-rebuild", "open": "yes", "author": "David Villalon / Maisa", "repo": "https://github.com/Davipar/djev-dev", "licence": "Apache-2.0", "underlying": "google/diffusiongemma-26b-a4b-it, BF16; full generation with thinking enabled", "has_distribution": true, "probability_source": [ "verbalized" ], "endpoint_condition": "our GPU (lium.io H200 141 GB), reached over the internet from Germany; serial, one request at a time", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 0.9583333333333334, "standard": 0.9895833333333334, "judge": 0.8013698630136986, "hard": 0.7772727272727272 }, "axes": { "intelligence": 80.83072059431514, "calibration": 92.70437317469347, "speed": 75.15117536916611, "cost": 26.85351129959409 }, "jevbench_score": 62.35960888483864, "speed": { "p50_s_raw": 0.4260098780505359, "p95_s_raw": 1.448969176481475, "p50_s_adjusted": 1.0020197561010717, "p95_s_adjusted": 3.04793835296295, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 1.061047127470374, "hard_tier_p95_s": 4.011991968052462 }, "cost": { "kind": "estimate", "usd_per_1000": 0.27429398876404487, "basis": "ESTIMATE: same-size hosted reference x 749 measured input and 690 measured output tokens per attempted decision across all 534, failures included", "usd_per_1000_v11_tiers": 0.159790127388535, "usd_per_1000_hard": 0.43772222727272725, "self_host_sensitivity": "Actual serial H200 rental equivalent: $0.846/1,000 decisions at measured 1.015s mean wall time and $3.00/hour; excludes idle/setup time." }, "calibration": { "score": 92.70437317469347, "score_label_only_as_onehot": null, "ece_hard": 0.06536279677768825, "probability_fidelity": 98.4813057049246, "brier_hard": 0.10142847386902532, "brier_standard_judge_v11": 0.020839209052624045, "note": null }, "hard": { "run": "runs/djev-thinking--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 178, "n_attempted": 220, "coverage": 1.0, "success_rate": 0.8090909090909091, "accuracy": 0.7772727272727272, "by_family": { "adversarial": { "correct": 11, "n": 12, "accuracy": 0.9166666666666666 }, "ambiguous": { "correct": 7, "n": 14, "accuracy": 0.5 }, "judge_hard": { "correct": 30, "n": 33, "accuracy": 0.9090909090909091 }, "long_policy": { "correct": 23, "n": 38, "accuracy": 0.6052631578947368 }, "multi_hop": { "correct": 26, "n": 35, "accuracy": 0.7428571428571429 }, "probability": { "correct": 19, "n": 20, "accuracy": 0.95 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 21, "n": 30, "accuracy": 0.7 }, "tradeoff": { "correct": 10, "n": 12, "accuracy": 0.8333333333333334 }, "trap": { "correct": 14, "n": 16, "accuracy": 0.875 } }, "has_distribution": true, "brier_mean": 0.10142847386902532, "ece": 0.06536279677768825, "probability_fidelity": 98.4813057049246, "calibration_score": 92.70437317469347, "onehot": { "ece": 0.0393258426966292, "probability_fidelity": 65.78526315789472, "calibration_score": 78.96004730928445 }, "latency_p50_s": 1.061047127470374, "latency_p95_s": 4.011991968052462, "mean_input_tokens": 1249.5045454545455, "mean_output_tokens": 1084.2227272727273, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 62.35960888483864, "Balanced 33:33:33 (no calibration)": 54.63921312233926, "Emphasis on Accuracy 60:20:20": 63.904764976472116, "Emphasis on Speed 20:60:20": 62.06931788837691, "Emphasis on Cost 20:20:60": 41.124733270617526, "Intelligence only": 80.83072059431514 }, "rank": 21, "rank_under": { "JevBench Score (25:25:25:25)": 21, "Balanced 33:33:33 (no calibration)": 30, "Emphasis on Accuracy 60:20:20": 25, "Emphasis on Speed 20:60:20": 27, "Emphasis on Cost 20:20:60": 34, "Intelligence only": 12 } }, { "key": "jev-local", "display": "jev-local (Qwen3.5-9B)", "class": "jev-rebuild", "open": "yes", "author": "us (GitHub)", "repo": "https://github.com/us/jev-local", "licence": "no licence stated in the repository (public code); Apache-2.0 base weights", "underlying": "Qwen/Qwen3.5-9B, frozen, per-option mean log-probability", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our RunPod GPU (H100 NVL 96 GB, Canada), reached over the internet from Germany", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.84375, "judge": 0.8904109589041096, "hard": 0.5909090909090909 }, "axes": { "intelligence": 70.76681508843012, "calibration": 68.7437159090909, "speed": 69.18842978642493, "cost": 43.327939808722704 }, "jevbench_score": 61.79680786600313, "speed": { "p50_s_raw": 1.0450106970965862, "p95_s_raw": 2.6157593585550782, "p50_s_adjusted": 2.2400213941931724, "p95_s_adjusted": 5.381518717110157, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 1.3785462081432343, "hard_tier_p95_s": 3.9017020471394064 }, "cost": { "kind": "estimate", "usd_per_1000": 0.07745842696629213, "basis": "ESTIMATE: hosted-provider price, OpenRouter qwen/qwen3.5-9b list price $0.1/M in, $0.0/M out (the exact base weights; scored by log-probabilities, nothing is generated) x 452 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts)", "usd_per_1000_v11_tiers": 0.0452, "usd_per_1000_hard": 0.1235, "self_host_sensitivity": null }, "calibration": { "score": 68.7437159090909, "score_label_only_as_onehot": null, "ece_hard": 0.14593909090909096, "probability_fidelity": 66.67524999999999, "brier_hard": 0.5767452409090912, "brier_standard_judge_v11": 0.1765305730578512, "note": null }, "hard": { "run": "runs/jev-local--hard (job jevbench-add-requests-20260919, run 4)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.5909090909090909, "by_family": { "adversarial": { "correct": 12, "n": 12, "accuracy": 1.0 }, "ambiguous": { "correct": 8, "n": 14, "accuracy": 0.5714285714285714 }, "judge_hard": { "correct": 23, "n": 33, "accuracy": 0.696969696969697 }, "long_policy": { "correct": 17, "n": 38, "accuracy": 0.4473684210526316 }, "multi_hop": { "correct": 20, "n": 35, "accuracy": 0.5714285714285714 }, "probability": { "correct": 8, "n": 20, "accuracy": 0.4 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 8, "n": 30, "accuracy": 0.26666666666666666 }, "tradeoff": { "correct": 9, "n": 12, "accuracy": 0.75 }, "trap": { "correct": 15, "n": 16, "accuracy": 0.9375 } }, "has_distribution": true, "brier_mean": 0.5767452409090912, "ece": 0.14593909090909096, "probability_fidelity": 66.67524999999999, "calibration_score": 68.7437159090909, "onehot": { "ece": 0.40909090909090906, "probability_fidelity": 39.6865, "calibration_score": 28.934159090909095 }, "latency_p50_s": 1.3785462081432343, "latency_p95_s": 3.9017020471394064, "mean_input_tokens": 636.8863636363636, "mean_output_tokens": 1.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 61.79680786600313, "Balanced 33:33:33 (no calibration)": 59.64083585655654, "Emphasis on Accuracy 60:20:20": 63.86429067808917, "Emphasis on Speed 20:60:20": 63.29065972079719, "Emphasis on Cost 20:20:60": 52.48478815240233, "Intelligence only": 70.76681508843012 }, "rank": 22, "rank_under": { "JevBench Score (25:25:25:25)": 22, "Balanced 33:33:33 (no calibration)": 22, "Emphasis on Accuracy 60:20:20": 26, "Emphasis on Speed 20:60:20": 26, "Emphasis on Cost 20:20:60": 21, "Intelligence only": 21 } }, { "key": "decider-2b", "display": "decider-2b (Mapika)", "class": "jev-rebuild", "open": "yes", "author": "Mapika", "repo": "https://huggingface.co/Mapika/decider-2b", "licence": "Apache-2.0", "underlying": "Qwen3.5-2B-Base with a trained decision readout, 1.9B", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our RunPod GPU (H100 NVL 96 GB, Canada), reached over the internet from Germany", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.8541666666666666, "judge": 0.773972602739726, "hard": 0.4727272727272727 }, "axes": { "intelligence": 61.24411705024726, "calibration": 46.58881818181821, "speed": 83.17774245885059, "cost": 60.99184724027941 }, "jevbench_score": 61.6816875223098, "speed": { "p50_s_raw": 0.26079703494906425, "p95_s_raw": 0.2831697516143322, "p50_s_adjusted": 0.6715940698981285, "p95_s_adjusted": 0.7163395032286645, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.2675044983625412, "hard_tier_p95_s": 0.45449532456696035 }, "cost": { "kind": "estimate", "usd_per_1000": 0.01996511235955056, "basis": "ESTIMATE: hosted-provider price, DeepInfra Qwen/Qwen3.5-4B list price $0.03/M in, $0.0/M out (no hosted ~2B Qwen3.5 is listed, so the 4B price is used and errs high; one pass, no output) x 312 input and 0 output tokens per decision (input tokens measured (the system's own count))", "usd_per_1000_v11_tiers": 0.00936076433121019, "usd_per_1000_hard": 0.035100409090909085, "self_host_sensitivity": null }, "calibration": { "score": 46.58881818181821, "score_label_only_as_onehot": null, "ece_hard": 0.3221518181818179, "probability_fidelity": 57.608000000000004, "brier_hard": 0.8062077985909092, "brier_standard_judge_v11": 0.2826622928099173, "note": null }, "hard": { "run": "runs/decider-2b--hard (job jevbench-add-requests-20260919, run 4)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.4727272727272727, "by_family": { "adversarial": { "correct": 9, "n": 12, "accuracy": 0.75 }, "ambiguous": { "correct": 4, "n": 14, "accuracy": 0.2857142857142857 }, "judge_hard": { "correct": 21, "n": 33, "accuracy": 0.6363636363636364 }, "long_policy": { "correct": 10, "n": 38, "accuracy": 0.2631578947368421 }, "multi_hop": { "correct": 17, "n": 35, "accuracy": 0.4857142857142857 }, "probability": { "correct": 6, "n": 20, "accuracy": 0.3 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 5, "n": 30, "accuracy": 0.16666666666666666 }, "tradeoff": { "correct": 6, "n": 12, "accuracy": 0.5 }, "trap": { "correct": 16, "n": 16, "accuracy": 1.0 } }, "has_distribution": true, "brier_mean": 0.8062077985909092, "ece": 0.3221518181818179, "probability_fidelity": 57.608000000000004, "calibration_score": 46.58881818181821, "onehot": { "ece": 0.5272727272727273, "probability_fidelity": 39.31099999999999, "calibration_score": 19.655499999999996 }, "latency_p50_s": 0.2675044983625412, "latency_p95_s": 0.45449532456696035, "mean_input_tokens": 1170.0136363636364, "mean_output_tokens": 0.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 61.6816875223098, "Balanced 33:33:33 (no calibration)": 67.73000347520238, "Emphasis on Accuracy 60:20:20": 65.05705651392121, "Emphasis on Speed 20:60:20": 73.53117574817735, "Emphasis on Cost 20:20:60": 64.94973350954271, "Intelligence only": 61.244117050247254 }, "rank": 23, "rank_under": { "JevBench Score (25:25:25:25)": 23, "Balanced 33:33:33 (no calibration)": 8, "Emphasis on Accuracy 60:20:20": 23, "Emphasis on Speed 20:60:20": 7, "Emphasis on Cost 20:20:60": 7, "Intelligence only": 30 } }, { "key": "nimble-9b", "display": "Bespoke Nimble 9B (Bespoke Labs)", "class": "jev-rebuild", "open": "yes", "author": "Bespoke Labs", "repo": "https://github.com/bespokelabsai/nimble", "licence": "Apache-2.0 (weights); repository without a licence file as of 19 Sep", "underlying": "bespokelabs/Bespoke-Nimble-9B (LoRA, adapter unchanged since 93ec5d6), merged into Qwen/Qwen3.5-9B@c202236 with the author's PEFT safe-merge", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our RunPod GPU (A40 48 GB, Canada), reached over the internet from Germany", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.9479166666666666, "judge": 0.8904109589041096, "hard": 0.6545454545454545 }, "axes": { "intelligence": 77.91214852943436, "calibration": 65.30347273546205, "speed": 78.68440871747626, "cost": 33.41009520749539 }, "jevbench_score": 60.47515137925396, "speed": { "p50_s_raw": 0.3889440931379795, "p95_s_raw": 0.6545137587934732, "p50_s_adjusted": 0.927888186275959, "p95_s_adjusted": 1.4590275175869463, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.444174911826849, "hard_tier_p95_s": 0.977684601396322 }, "cost": { "kind": "estimate", "usd_per_1000": 0.16583014981273408, "basis": "ESTIMATE: hosted-provider price, openrouter qwen/qwen3.5-9b list price $0.1/M in, $0.15/M out (a LoRA merge of Qwen3.5-9B; the base weights are listed on OpenRouter (size class dense_9B), as in the v1.1.3 row) x 970 input and 1 output tokens per decision (input tokens measured (the system's own count))", "usd_per_1000_v11_tiers": 0.09719840764331211, "usd_per_1000_hard": 0.26378636363636365, "self_host_sensitivity": null }, "calibration": { "score": 65.30347273546205, "score_label_only_as_onehot": null, "ece_hard": 0.19055518585790857, "probability_fidelity": 68.71798264250582, "brier_hard": 0.5272551742334669, "brier_standard_judge_v11": 0.1554427753099633, "note": null }, "hard": { "run": "runs/nimble-9b--hard-r4 (job jevbench-add-requests-20260919, run 4)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.6545454545454545, "by_family": { "adversarial": { "correct": 9, "n": 12, "accuracy": 0.75 }, "ambiguous": { "correct": 7, "n": 14, "accuracy": 0.5 }, "judge_hard": { "correct": 25, "n": 33, "accuracy": 0.7575757575757576 }, "long_policy": { "correct": 23, "n": 38, "accuracy": 0.6052631578947368 }, "multi_hop": { "correct": 28, "n": 35, "accuracy": 0.8 }, "probability": { "correct": 12, "n": 20, "accuracy": 0.6 }, "routing_hard": { "correct": 9, "n": 10, "accuracy": 0.9 }, "temporal_numeric": { "correct": 9, "n": 30, "accuracy": 0.3 }, "tradeoff": { "correct": 7, "n": 12, "accuracy": 0.5833333333333334 }, "trap": { "correct": 15, "n": 16, "accuracy": 0.9375 } }, "has_distribution": true, "brier_mean": 0.5272551742334669, "ece": 0.19055518585790857, "probability_fidelity": 68.71798264250582, "calibration_score": 65.30347273546205, "onehot": { "ece": 0.34545454545454546, "probability_fidelity": 51.72950000000001, "calibration_score": 41.319295454545454 }, "latency_p50_s": 0.444174911826849, "latency_p95_s": 0.977684601396322, "mean_input_tokens": 2636.3636363636365, "mean_output_tokens": 2.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 60.47515137925396, "Balanced 33:33:33 (no calibration)": 58.946387292930616, "Emphasis on Accuracy 60:20:20": 65.90470011684901, "Emphasis on Speed 20:60:20": 66.16522446328183, "Emphasis on Cost 20:20:60": 46.97052357201384, "Intelligence only": 77.91214852943433 }, "rank": 24, "rank_under": { "JevBench Score (25:25:25:25)": 24, "Balanced 33:33:33 (no calibration)": 24, "Emphasis on Accuracy 60:20:20": 20, "Emphasis on Speed 20:60:20": 22, "Emphasis on Cost 20:20:60": 25, "Intelligence only": 19 } }, { "key": "gemini-3.1-flash-lite", "display": "Gemini 3.1 Flash-Lite", "class": "llm-baseline", "open": "no", "author": "Google", "repo": null, "licence": "proprietary API", "underlying": "closed", "has_distribution": true, "probability_source": [ "verbalized" ], "endpoint_condition": "production API (Google)", "endpoint_kind": "api", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.9895833333333334, "judge": 0.9315068493150684, "hard": 0.75 }, "axes": { "intelligence": 85.56083319133411, "calibration": 68.08811278368097, "speed": 81.78762581772229, "cost": 27.362399077259013 }, "jevbench_score": 60.08928259753951, "speed": { "p50_s_raw": 0.756230715662241, "p95_s_raw": 0.8761593606323003, "p50_s_adjusted": 0.756230715662241, "p95_s_adjusted": 0.8761593606323003, "adjustment": "none (production API)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.7880580946803093, "hard_tier_p95_s": 1.4522175978869198 }, "cost": { "kind": "measured", "usd_per_1000": 0.26378698501872655, "basis": "public tariff x measured tokens (https://ai.google.dev/gemini-api/docs/pricing (paid tier, read 2026-09-19)) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | public tariff x measured tokens (hard-tier run)", "usd_per_1000_v11_tiers": 0.17806050955414016, "usd_per_1000_hard": 0.38614204545454534, "self_host_sensitivity": null }, "calibration": { "score": 68.08811278368097, "score_label_only_as_onehot": null, "ece_hard": 0.2667577285845467, "probability_fidelity": 89.52777128427128, "brier_hard": 0.5096290133859069, "brier_standard_judge_v11": 0.09465247933884294, "note": null }, "hard": { "run": "runs/gemini-3.1-flash-lite--hard", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.75, "by_family": { "adversarial": { "correct": 12, "n": 12, "accuracy": 1.0 }, "ambiguous": { "correct": 11, "n": 14, "accuracy": 0.7857142857142857 }, "judge_hard": { "correct": 29, "n": 33, "accuracy": 0.8787878787878788 }, "long_policy": { "correct": 25, "n": 38, "accuracy": 0.6578947368421053 }, "multi_hop": { "correct": 27, "n": 35, "accuracy": 0.7714285714285715 }, "probability": { "correct": 18, "n": 20, "accuracy": 0.9 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 9, "n": 30, "accuracy": 0.3 }, "tradeoff": { "correct": 9, "n": 12, "accuracy": 0.75 }, "trap": { "correct": 15, "n": 16, "accuracy": 0.9375 } }, "has_distribution": true, "brier_mean": 0.5096290133859069, "ece": 0.2667577285845467, "probability_fidelity": 89.52777128427128, "calibration_score": 68.08811278368097, "onehot": { "ece": 0.25, "probability_fidelity": 63.314499999999995, "calibration_score": 56.65725 }, "latency_p50_s": 0.7880580946803093, "latency_p95_s": 1.4522175978869198, "mean_input_tokens": 1234.5045454545455, "mean_output_tokens": 51.67727272727273, "charged_usd": 0.08495124999999998 }, "presets": { "JevBench Score (25:25:25:25)": 60.08928259753951, "Balanced 33:33:33 (no calibration)": 57.63756076318859, "Emphasis on Accuracy 60:20:20": 67.50459839724847, "Emphasis on Speed 20:60:20": 66.29768996811387, "Emphasis on Cost 20:20:60": 42.78435852846635, "Intelligence only": 85.56083319133413 }, "rank": 25, "rank_under": { "JevBench Score (25:25:25:25)": 25, "Balanced 33:33:33 (no calibration)": 25, "Emphasis on Accuracy 60:20:20": 15, "Emphasis on Speed 20:60:20": 20, "Emphasis on Cost 20:20:60": 32, "Intelligence only": 6 } }, { "key": "openjev-thinking", "display": "OpenJev (thinking, BF16)", "class": "jev-rebuild", "open": "yes", "author": "razorback16", "repo": "https://github.com/razorback16/openjev", "licence": "Apache-2.0", "underlying": "google/diffusiongemma-26b-a4b-it, BF16; OpenJev think=512", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our GPU (lium.io H200 141 GB), reached over the internet from Germany; serial, one request at a time", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 1.0, "judge": 0.9452054794520548, "hard": 0.7818181818181819 }, "axes": { "intelligence": 87.96811832253644, "calibration": 69.55970870533434, "speed": 76.05530411890852, "cost": 27.82225297231227 }, "jevbench_score": 59.98618116922599, "speed": { "p50_s_raw": 0.46296589844860137, "p95_s_raw": 1.0775369299459272, "p50_s_adjusted": 1.0759317968972026, "p95_s_adjusted": 2.3050738598918543, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.8800784219056368, "hard_tier_p95_s": 1.2348620565375312 }, "cost": { "kind": "estimate", "usd_per_1000": 0.25463898876404495, "basis": "ESTIMATE: same hosted reference x 1778 billed input and 315 thought output tokens per decision", "usd_per_1000_v11_tiers": 0.15592996815286625, "usd_per_1000_hard": 0.39552368181818176, "self_host_sensitivity": "Actual serial H200 rental equivalent: $0.585/1,000 decisions at measured 0.702s mean wall time and $3.00/hour; excludes setup/idle time." }, "calibration": { "score": 69.55970870533434, "score_label_only_as_onehot": null, "ece_hard": 0.13293071724492195, "probability_fidelity": 65.70556085965308, "brier_hard": 0.3361950043526198, "brier_standard_judge_v11": 0.06627872347289883, "note": null }, "hard": { "run": "runs/openjev-thinking--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.7818181818181819, "by_family": { "adversarial": { "correct": 12, "n": 12, "accuracy": 1.0 }, "ambiguous": { "correct": 14, "n": 14, "accuracy": 1.0 }, "judge_hard": { "correct": 31, "n": 33, "accuracy": 0.9393939393939394 }, "long_policy": { "correct": 19, "n": 38, "accuracy": 0.5 }, "multi_hop": { "correct": 29, "n": 35, "accuracy": 0.8285714285714286 }, "probability": { "correct": 15, "n": 20, "accuracy": 0.75 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 14, "n": 30, "accuracy": 0.4666666666666667 }, "tradeoff": { "correct": 12, "n": 12, "accuracy": 1.0 }, "trap": { "correct": 16, "n": 16, "accuracy": 1.0 } }, "has_distribution": true, "brier_mean": 0.3361950043526198, "ece": 0.13293071724492195, "probability_fidelity": 65.70556085965308, "calibration_score": 69.55970870533434, "onehot": { "ece": 0.21818181818181814, "probability_fidelity": 55.79700000000001, "calibration_score": 56.08031818181819 }, "latency_p50_s": 0.8800784219056368, "latency_p95_s": 1.2348620565375312, "mean_input_tokens": 2901.8136363636363, "mean_output_tokens": 447.8681818181818, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 59.98618116922599, "Balanced 33:33:33 (no calibration)": 57.097317040809784, "Emphasis on Accuracy 60:20:20": 67.87339013833852, "Emphasis on Speed 20:60:20": 64.03556176946387, "Emphasis on Cost 20:20:60": 42.82785651242419, "Intelligence only": 87.96811832253647 }, "rank": 26, "rank_under": { "JevBench Score (25:25:25:25)": 26, "Balanced 33:33:33 (no calibration)": 27, "Emphasis on Accuracy 60:20:20": 13, "Emphasis on Speed 20:60:20": 25, "Emphasis on Cost 20:20:60": 31, "Intelligence only": 3 } }, { "key": "kev-4b", "display": "kev 4B (research preview)", "class": "jev-rebuild", "open": true, "author": "Jared Palmer", "repo": "https://github.com/jaredpalmer/kev", "licence": "Apache-2.0", "underlying": "Qwen3-4B-Base + LoRA + learned pointer head; jaredpalmer/kev-4b", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our RunPod GPU (GeForce RTX 3090 24 GB, community cloud CA), reached over the internet", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.9166666666666666, "judge": 0.8561643835616438, "hard": 0.42272727272727273 }, "axes": { "intelligence": 64.79617353902547, "calibration": 41.97093961668896, "speed": 75.73845187472214, "cost": 61.77327904537152 }, "jevbench_score": 59.724673285425375, "speed": { "p50_s_raw": 0.5502474755048752, "p95_s_raw": 0.9917014226317405, "p50_s_adjusted": 1.2504949510097503, "p95_s_adjusted": 2.1334028452634812, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.718296229839325, "hard_tier_p95_s": 1.7924389142543065 }, "cost": { "kind": "estimate", "usd_per_1000": 0.018802865168539323, "basis": "ESTIMATE: hosted-provider price, DeepInfra Qwen3.5-4B size-class reference list price $0.03/M in, $0.0/M out (a 4B one-pass model with no generated output) x 279 input and 0 output tokens per decision (input tokens measured (the system's own count))", "usd_per_1000_v11_tiers": 0.008382324840764331, "usd_per_1000_hard": 0.033675818181818175, "self_host_sensitivity": null }, "calibration": { "score": 41.97093961668896, "score_label_only_as_onehot": null, "ece_hard": 0.40195757303003016, "probability_fidelity": 64.33339383938394, "brier_hard": 0.8860972000262979, "brier_standard_judge_v11": 0.17123441707403728, "note": null }, "hard": { "run": "runs/kev-4b--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.42272727272727273, "by_family": { "adversarial": { "correct": 9, "n": 12, "accuracy": 0.75 }, "ambiguous": { "correct": 2, "n": 14, "accuracy": 0.14285714285714285 }, "judge_hard": { "correct": 18, "n": 33, "accuracy": 0.5454545454545454 }, "long_policy": { "correct": 9, "n": 38, "accuracy": 0.23684210526315788 }, "multi_hop": { "correct": 16, "n": 35, "accuracy": 0.45714285714285713 }, "probability": { "correct": 9, "n": 20, "accuracy": 0.45 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 3, "n": 30, "accuracy": 0.1 }, "tradeoff": { "correct": 3, "n": 12, "accuracy": 0.25 }, "trap": { "correct": 14, "n": 16, "accuracy": 0.875 } }, "has_distribution": true, "brier_mean": 0.8860972000262979, "ece": 0.40195757303003016, "probability_fidelity": 64.33339383938394, "calibration_score": 41.97093961668896, "onehot": { "ece": 0.5772727272727273, "probability_fidelity": 45.35549999999999, "calibration_score": 22.677749999999996 }, "latency_p50_s": 0.718296229839325, "latency_p95_s": 1.7924389142543065, "mean_input_tokens": 1122.5272727272727, "mean_output_tokens": 61.42727272727273, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 59.724673285425375, "Balanced 33:33:33 (no calibration)": 67.17723837793268, "Emphasis on Accuracy 60:20:20": 66.21448898508923, "Emphasis on Speed 20:60:20": 70.47902009536537, "Emphasis on Cost 20:20:60": 64.96112685084904, "Intelligence only": 64.79617353902549 }, "rank": 27, "rank_under": { "JevBench Score (25:25:25:25)": 27, "Balanced 33:33:33 (no calibration)": 10, "Emphasis on Accuracy 60:20:20": 18, "Emphasis on Speed 20:60:20": 12, "Emphasis on Cost 20:20:60": 6, "Intelligence only": 25 } }, { "key": "deepseek-flash", "display": "DeepSeek V4.1 Flash (thinking default)", "class": "llm-baseline", "open": "weights", "author": "DeepSeek", "repo": null, "licence": "open weights, proprietary API route", "underlying": "DeepSeek-V4.1-Flash", "has_distribution": true, "probability_source": [ "verbalized" ], "endpoint_condition": "production API (DeepSeek)", "endpoint_kind": "api", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 0.9861111111111112, "standard": 0.9895833333333334, "judge": 0.9315068493150684, "hard": 0.95 }, "axes": { "intelligence": 94.33138029881806, "calibration": 96.66517503108454, "speed": 71.59823070734842, "cost": 16.793370728606376 }, "jevbench_score": 57.54288064128547, "speed": { "p50_s_raw": 1.4163860343396664, "p95_s_raw": 4.886470635980367, "p50_s_adjusted": 1.4163860343396664, "p95_s_adjusted": 4.886470635980367, "adjustment": "none (production API)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 3.1479117795825005, "hard_tier_p95_s": 27.954364685714225 }, "cost": { "kind": "measured", "usd_per_1000": 0.593682584269663, "basis": "public tariff x measured tokens (https://api-docs.deepseek.com/quick_start/pricing (cache-miss off-peak; the run is on a Saturday, off-peak all day)) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | public tariff x measured tokens (hard-tier run)", "usd_per_1000_v11_tiers": 0.2505138535031847, "usd_per_1000_hard": 1.083477954545455, "self_host_sensitivity": null }, "calibration": { "score": 96.66517503108454, "score_label_only_as_onehot": null, "ece_hard": 0.03334125816558947, "probability_fidelity": 99.998601695287, "brier_hard": 0.043915329620841076, "brier_standard_judge_v11": 0.02778397179722692, "note": null }, "hard": { "run": "runs/deepseek-flash--hard-mt16k", "n_items": 220, "n_ok": 212, "n_attempted": 220, "coverage": 1.0, "success_rate": 0.9636363636363636, "accuracy": 0.95, "by_family": { "adversarial": { "correct": 12, "n": 12, "accuracy": 1.0 }, "ambiguous": { "correct": 13, "n": 14, "accuracy": 0.9285714285714286 }, "judge_hard": { "correct": 30, "n": 33, "accuracy": 0.9090909090909091 }, "long_policy": { "correct": 34, "n": 38, "accuracy": 0.8947368421052632 }, "multi_hop": { "correct": 35, "n": 35, "accuracy": 1.0 }, "probability": { "correct": 19, "n": 20, "accuracy": 0.95 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 28, "n": 30, "accuracy": 0.9333333333333333 }, "tradeoff": { "correct": 12, "n": 12, "accuracy": 1.0 }, "trap": { "correct": 16, "n": 16, "accuracy": 1.0 } }, "has_distribution": true, "brier_mean": 0.043915329620841076, "ece": 0.03334125816558947, "probability_fidelity": 99.998601695287, "calibration_score": 96.66517503108454, "onehot": { "ece": 0.014150943396226467, "probability_fidelity": 65.48263157894736, "calibration_score": 81.32622144985103 }, "latency_p50_s": 3.1479117795825005, "latency_p95_s": 27.954364685714225, "mean_input_tokens": 1189.15, "mean_output_tokens": 1752.8727272727272, "charged_usd": 0.23836515000000008 }, "presets": { "JevBench Score (25:25:25:25)": 57.54288064128547, "Balanced 33:33:33 (no calibration)": 48.40595414642161, "Emphasis on Accuracy 60:20:20": 63.212322443041295, "Emphasis on Speed 20:60:20": 56.61091666630342, "Emphasis on Cost 20:20:60": 31.695267485021212, "Intelligence only": 94.33138029881802 }, "rank": 28, "rank_under": { "JevBench Score (25:25:25:25)": 28, "Balanced 33:33:33 (no calibration)": 34, "Emphasis on Accuracy 60:20:20": 28, "Emphasis on Speed 20:60:20": 33, "Emphasis on Cost 20:20:60": 40, "Intelligence only": 2 } }, { "key": "kev-8b", "display": "kev 8B (research preview)", "class": "jev-rebuild", "open": true, "author": "Jared Palmer", "repo": "https://github.com/jaredpalmer/kev", "licence": "Apache-2.0", "underlying": "Qwen3-8B-Base + LoRA + learned pointer head; jaredpalmer/kev-8b", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our RunPod GPU (GeForce RTX 3090 24 GB, community cloud CA), reached over the internet", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.9270833333333334, "judge": 0.9041095890410958, "hard": 0.4727272727272727 }, "axes": { "intelligence": 69.38030902507023, "calibration": 44.18419951086019, "speed": 74.85933162711135, "cost": 44.04134083457654 }, "jevbench_score": 56.383551097366656, "speed": { "p50_s_raw": 0.5904169715940952, "p95_s_raw": 1.1521932914853092, "p50_s_adjusted": 1.3308339431881904, "p95_s_adjusted": 2.4543865829706184, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.7628460973501205, "hard_tier_p95_s": 2.0858752064406856 }, "cost": { "kind": "estimate", "usd_per_1000": 0.07333117415730338, "basis": "ESTIMATE: hosted-provider price, OpenRouter qwen/qwen3-8b list price list price $0.117/M in, $0.0/M out (the same-size Qwen3-8B weights; kev generates no output tokens) x 279 input and 0 output tokens per decision (input tokens measured (the system's own count))", "usd_per_1000_v11_tiers": 0.03269106687898089, "usd_per_1000_hard": 0.13133569090909092, "self_host_sensitivity": null }, "calibration": { "score": 44.18419951086019, "score_label_only_as_onehot": null, "ece_hard": 0.36420177926883585, "probability_fidelity": 61.20875487548755, "brier_hard": 0.8373365943511044, "brier_standard_judge_v11": 0.13167773335725774, "note": null }, "hard": { "run": "runs/kev-8b--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.4727272727272727, "by_family": { "adversarial": { "correct": 8, "n": 12, "accuracy": 0.6666666666666666 }, "ambiguous": { "correct": 3, "n": 14, "accuracy": 0.21428571428571427 }, "judge_hard": { "correct": 20, "n": 33, "accuracy": 0.6060606060606061 }, "long_policy": { "correct": 15, "n": 38, "accuracy": 0.39473684210526316 }, "multi_hop": { "correct": 18, "n": 35, "accuracy": 0.5142857142857142 }, "probability": { "correct": 10, "n": 20, "accuracy": 0.5 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 0, "n": 30, "accuracy": 0.0 }, "tradeoff": { "correct": 5, "n": 12, "accuracy": 0.4166666666666667 }, "trap": { "correct": 15, "n": 16, "accuracy": 0.9375 } }, "has_distribution": true, "brier_mean": 0.8373365943511044, "ece": 0.36420177926883585, "probability_fidelity": 61.20875487548755, "calibration_score": 44.18419951086019, "onehot": { "ece": 0.5272727272727273, "probability_fidelity": 47.43100000000001, "calibration_score": 23.715500000000006 }, "latency_p50_s": 0.7628460973501205, "latency_p95_s": 2.0858752064406856, "mean_input_tokens": 1122.5272727272727, "mean_output_tokens": 61.21363636363636, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 56.383551097366656, "Balanced 33:33:33 (no calibration)": 61.157196567349956, "Emphasis on Accuracy 60:20:20": 64.32251745536995, "Emphasis on Speed 20:60:20": 66.30815128100333, "Emphasis on Cost 20:20:60": 53.63061220562486, "Intelligence only": 69.38030902507025 }, "rank": 29, "rank_under": { "JevBench Score (25:25:25:25)": 29, "Balanced 33:33:33 (no calibration)": 20, "Emphasis on Accuracy 60:20:20": 24, "Emphasis on Speed 20:60:20": 19, "Emphasis on Cost 20:20:60": 19, "Intelligence only": 24 } }, { "key": "open-jev-zefan-9b", "display": "Open-Jev 9B (Zefan Cai)", "class": "jev-rebuild", "open": "yes", "author": "Zefan Cai (@Zefan_Cai)", "repo": "https://github.com/Zefan-Cai/Open-Jev", "licence": "MIT (loader); Apache-2.0 (adapter and pinned Qwen base); CC0-1.0 public training projection", "underlying": "Qwen3.5-9B plus rank-8 LoRA and trained scalar decision head", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our RunPod GPU (H100 80GB HBM3), reached over the internet", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.90625, "judge": 0.815068493150685, "hard": 0.6090909090909091 }, "axes": { "intelligence": 71.16915718206087, "calibration": 63.2782433598038, "speed": 72.03286755510636, "cost": 28.123361021000306 }, "jevbench_score": 54.95864804501749, "speed": { "p50_s_raw": 0.754859171807766, "p95_s_raw": 1.8114654466509819, "p50_s_adjusted": 1.6597183436155318, "p95_s_adjusted": 3.7729308933019636, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.7952907048165798, "hard_tier_p95_s": 1.648427233844994 }, "cost": { "kind": "estimate", "usd_per_1000": 0.24882153558052428, "basis": "ESTIMATE: hosted-provider price, OpenRouter Qwen3.5-9B list price read 2026-09-21 list price $0.1/M in, $0.0/M out (the exact 9B base and a conservative same-family proxy for the unlisted 2B; the decision head generates no output tokens) x 1439 input and 0 output tokens per decision (input tokens measured (the system's own count))", "usd_per_1000_v11_tiers": 0.14393662420382164, "usd_per_1000_hard": 0.39852090909090904, "self_host_sensitivity": null }, "calibration": { "score": 63.2782433598038, "score_label_only_as_onehot": null, "ece_hard": 0.19026066635697753, "probability_fidelity": 64.60861999100311, "brier_hard": 0.5546258172920748, "brier_standard_judge_v11": 0.23654785588059785, "note": null }, "hard": { "run": "runs/open-jev-zefan-9b--hard (job jevbench-zefan-openjev-20260921)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.6090909090909091, "by_family": { "adversarial": { "correct": 11, "n": 12, "accuracy": 0.9166666666666666 }, "ambiguous": { "correct": 7, "n": 14, "accuracy": 0.5 }, "judge_hard": { "correct": 27, "n": 33, "accuracy": 0.8181818181818182 }, "long_policy": { "correct": 17, "n": 38, "accuracy": 0.4473684210526316 }, "multi_hop": { "correct": 24, "n": 35, "accuracy": 0.6857142857142857 }, "probability": { "correct": 10, "n": 20, "accuracy": 0.5 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 6, "n": 30, "accuracy": 0.2 }, "tradeoff": { "correct": 6, "n": 12, "accuracy": 0.5 }, "trap": { "correct": 16, "n": 16, "accuracy": 1.0 } }, "has_distribution": true, "brier_mean": 0.5546258172920748, "ece": 0.19026066635697753, "probability_fidelity": 64.60861999100311, "calibration_score": 63.2782433598038, "onehot": { "ece": 0.3909090909090909, "probability_fidelity": 46.07800000000001, "calibration_score": 33.94809090909092 }, "latency_p50_s": 0.7952907048165798, "latency_p95_s": 1.648427233844994, "mean_input_tokens": 3985.2090909090907, "mean_output_tokens": 0.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 54.95864804501749, "Balanced 33:33:33 (no calibration)": 52.436043637380706, "Emphasis on Accuracy 60:20:20": 59.25086279742271, "Emphasis on Speed 20:60:20": 59.53745020678049, "Emphasis on Cost 20:20:60": 40.87001889639716, "Intelligence only": 71.16915718206087 }, "rank": 30, "rank_under": { "JevBench Score (25:25:25:25)": 30, "Balanced 33:33:33 (no calibration)": 32, "Emphasis on Accuracy 60:20:20": 31, "Emphasis on Speed 20:60:20": 30, "Emphasis on Cost 20:20:60": 35, "Intelligence only": 20 } }, { "key": "system-one-sg", "display": "system-one (Qwen3-8B, Sean Goedecke)", "class": "jev-rebuild", "open": "yes", "author": "Sean Goedecke", "repo": "https://github.com/sgoedecke/system-one", "licence": "no licence file in the repository as of 19 Sep; Qwen3 weights Apache-2.0", "underlying": "Qwen/Qwen3-8B (frozen, BF16), the model of the author's demos", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our RunPod GPU (RTX PRO 4500 Blackwell 32 GB (EU-RO-1)), reached over the internet", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.90625, "judge": 0.9178082191780822, "hard": 0.5 }, "axes": { "intelligence": 70.3016034401033, "calibration": 36.764309957143894, "speed": 84.3621367799104, "cost": 41.45387606228526 }, "jevbench_score": 54.830990505255514, "speed": { "p50_s_raw": 0.1661309413611889, "p95_s_raw": 0.30472867079079147, "p50_s_adjusted": 0.4822618827223778, "p95_s_adjusted": 0.759457341581583, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": "RunPod RTX PRO 4500 Blackwell 32 GB (EU-RO-1)", "measured_where": "from a Hetzner server in Germany over the internet to the pod's public TCP port (plain HTTP, one connection per request) through a thin transport around the author's library; model loaded before timing", "hard_tier_p50_s": 0.20696432143449783, "hard_tier_p95_s": 0.65277008600533 }, "cost": { "kind": "estimate", "usd_per_1000": 0.08944116853932584, "basis": "ESTIMATE: hosted-provider price, openrouter qwen/qwen3-8b list price $0.117/M in, $0.455/M out (same weights, listed on OpenRouter) x 412 input and 1 output tokens per decision (input tokens measured) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | ESTIMATE: openrouter qwen/qwen3-8b $0.117/M in, $0.455/M out x 1258 in / 1 out tokens per hard decision", "usd_per_1000_v11_tiers": 0.04869626114649681, "usd_per_1000_hard": 0.14759526363636366, "self_host_sensitivity": { "usd_per_1000": 0.030492280369226854, "score": 62.895252391243304, "machine": "1x RTX PRO 4500 Blackwell 32 GB (EU-RO-1) (on-demand, RunPod secure)", "usd_per_h": 0.72, "concurrency": 4, "utilisation": 0.3, "p50_s_used": 0.18295368221536112, "decisions_per_hour": 23612.533771879913 } }, "calibration": { "score": 36.764309957143894, "score_label_only_as_onehot": null, "ece_hard": 0.42428537094007235, "probability_fidelity": 58.38569410230225, "brier_hard": 0.9241372110675431, "brier_standard_judge_v11": 0.1602412905726563, "note": null }, "hard": { "run": "runs-gpu/system-one-sg--hard", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.5, "by_family": { "adversarial": { "correct": 8, "n": 12, "accuracy": 0.6666666666666666 }, "ambiguous": { "correct": 5, "n": 14, "accuracy": 0.35714285714285715 }, "judge_hard": { "correct": 20, "n": 33, "accuracy": 0.6060606060606061 }, "long_policy": { "correct": 12, "n": 38, "accuracy": 0.3157894736842105 }, "multi_hop": { "correct": 17, "n": 35, "accuracy": 0.4857142857142857 }, "probability": { "correct": 10, "n": 20, "accuracy": 0.5 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 8, "n": 30, "accuracy": 0.26666666666666666 }, "tradeoff": { "correct": 5, "n": 12, "accuracy": 0.4166666666666667 }, "trap": { "correct": 15, "n": 16, "accuracy": 0.9375 } }, "has_distribution": true, "brier_mean": 0.9241372110675431, "ece": 0.42428537094007235, "probability_fidelity": 58.38569410230225, "calibration_score": 36.764309957143894, "onehot": { "ece": 0.5, "probability_fidelity": 45.96800000000001, "calibration_score": 22.984000000000005 }, "latency_p50_s": 0.20696432143449783, "latency_p95_s": 0.65277008600533, "mean_input_tokens": 1257.6090909090908, "mean_output_tokens": 1.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 54.830990505255514, "Balanced 33:33:33 (no calibration)": 62.64589941235941, "Emphasis on Accuracy 60:20:20": 65.60269510128042, "Emphasis on Speed 20:60:20": 70.56585660083952, "Emphasis on Cost 20:20:60": 53.10820795447075, "Intelligence only": 70.30160344010334 }, "rank": 31, "rank_under": { "JevBench Score (25:25:25:25)": 31, "Balanced 33:33:33 (no calibration)": 17, "Emphasis on Accuracy 60:20:20": 21, "Emphasis on Speed 20:60:20": 11, "Emphasis on Cost 20:20:60": 20, "Intelligence only": 22 } }, { "key": "jeff", "display": "jeff (Logan Markewich, GLiFormer 400M)", "class": "jev-rebuild", "open": "yes", "author": "Logan Markewich", "repo": "https://github.com/logan-markewich/jeff", "licence": "MIT (code); GLiFormer weights per their model card", "underlying": "GLiFormer large (knowledgator/gliformer-large-v1, ~400M) behind a TypeSafe-compatible /v1/systemone server", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our CPU (4 threads, Ryzen 5 3600)", "endpoint_kind": "cpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.7604166666666666, "judge": 0.6164383561643836, "hard": 0.37727272727272726 }, "axes": { "intelligence": 46.854834433980876, "calibration": 64.5620340909091, "speed": 63.49232389662579, "cost": 76.57596288088385 }, "jevbench_score": 54.38199454750733, "speed": { "p50_s_raw": 0.9379300177097321, "p95_s_raw": 10.969045254960655, "p50_s_adjusted": 2.025860035419464, "p95_s_adjusted": 22.088090509921308, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads", "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing", "hard_tier_p50_s": 2.2362988367676735, "hard_tier_p95_s": 55.099180381745 }, "cost": { "kind": "estimate", "usd_per_1000": 0.006036722846441948, "basis": "ESTIMATE: hosted-provider price, deepinfra encoders of the same size (bge-large, e5-large, Qwen3-Embedding-0.6B) list price $0.01/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 272 input and 0 output tokens per decision (input tokens measured (the system's own count))", "usd_per_1000_v11_tiers": 0.0027223885350318475, "usd_per_1000_hard": 0.010767181818181818, "self_host_sensitivity": null }, "calibration": { "score": 64.5620340909091, "score_label_only_as_onehot": null, "ece_hard": 0.18531090909090905, "probability_fidelity": 66.18625, "brier_hard": 0.7454531705909093, "brier_standard_judge_v11": 0.5159533382231405, "note": null }, "hard": { "run": "runs/jeff--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.37727272727272726, "by_family": { "adversarial": { "correct": 7, "n": 12, "accuracy": 0.5833333333333334 }, "ambiguous": { "correct": 0, "n": 14, "accuracy": 0.0 }, "judge_hard": { "correct": 17, "n": 33, "accuracy": 0.5151515151515151 }, "long_policy": { "correct": 5, "n": 38, "accuracy": 0.13157894736842105 }, "multi_hop": { "correct": 17, "n": 35, "accuracy": 0.4857142857142857 }, "probability": { "correct": 7, "n": 20, "accuracy": 0.35 }, "routing_hard": { "correct": 4, "n": 10, "accuracy": 0.4 }, "temporal_numeric": { "correct": 10, "n": 30, "accuracy": 0.3333333333333333 }, "tradeoff": { "correct": 5, "n": 12, "accuracy": 0.4166666666666667 }, "trap": { "correct": 11, "n": 16, "accuracy": 0.6875 } }, "has_distribution": true, "brier_mean": 0.7454531705909093, "ece": 0.18531090909090905, "probability_fidelity": 66.18625, "calibration_score": 64.5620340909091, "onehot": { "ece": 0.6227272727272728, "probability_fidelity": 37.17100000000001, "calibration_score": 18.585500000000003 }, "latency_p50_s": 2.2362988367676735, "latency_p95_s": 55.099180381745, "mean_input_tokens": 1076.7181818181818, "mean_output_tokens": 6.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 54.38199454750733, "Balanced 33:33:33 (no calibration)": 53.63210450784728, "Emphasis on Accuracy 60:20:20": 48.23743306851332, "Emphasis on Speed 20:60:20": 54.471698224671705, "Emphasis on Cost 20:20:60": 58.710991200343976, "Intelligence only": 41.145582413508365 }, "rank": 32, "rank_under": { "JevBench Score (25:25:25:25)": 32, "Balanced 33:33:33 (no calibration)": 31, "Emphasis on Accuracy 60:20:20": 33, "Emphasis on Speed 20:60:20": 34, "Emphasis on Cost 20:20:60": 13, "Intelligence only": 33 } }, { "key": "laya", "display": "Laya (Convai Innovations, ModernBERT-large 421M)", "class": "jev-rebuild", "open": "yes", "author": "Convai Innovations", "repo": "https://huggingface.co/convaiinnovations/laya", "licence": "Apache-2.0", "underlying": "ModernBERT-large encoder + option-marker decision head, 421M, RLCD-trained", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our CPU (4 threads, Ryzen 5 3600)", "endpoint_kind": "cpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 0.9444444444444444, "standard": 0.7291666666666666, "judge": 0.6917808219178082, "hard": 0.3409090909090909 }, "axes": { "intelligence": 45.82464454274022, "calibration": 62.46471590909091, "speed": 71.05966297458822, "cost": 86.20408526325653 }, "jevbench_score": 54.35373809070198, "speed": { "p50_s_raw": 0.787067785859108, "p95_s_raw": 2.197125389799475, "p50_s_adjusted": 1.7241355717182159, "p95_s_adjusted": 4.544250779598951, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads", "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing", "hard_tier_p50_s": 1.9288412630558014, "hard_tier_p95_s": 2.2888929322361946 }, "cost": { "kind": "estimate", "usd_per_1000": 0.0028831273408239703, "basis": "ESTIMATE: hosted-provider price, deepinfra encoders of the same size (bge-large, e5-large, Qwen3-Embedding-0.6B) list price $0.01/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 205 input and 0 output tokens per decision (input tokens measured (the system's own count))", "usd_per_1000_v11_tiers": 0.002048248407643312, "usd_per_1000_hard": 0.004074727272727273, "self_host_sensitivity": null }, "calibration": { "score": 62.46471590909091, "score_label_only_as_onehot": null, "ece_hard": 0.20550909090909092, "probability_fidelity": 66.03125, "brier_hard": 0.7660523635454545, "brier_standard_judge_v11": 0.41431437239669405, "note": null }, "hard": { "run": "runs/laya--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.3409090909090909, "by_family": { "adversarial": { "correct": 8, "n": 12, "accuracy": 0.6666666666666666 }, "ambiguous": { "correct": 2, "n": 14, "accuracy": 0.14285714285714285 }, "judge_hard": { "correct": 12, "n": 33, "accuracy": 0.36363636363636365 }, "long_policy": { "correct": 12, "n": 38, "accuracy": 0.3157894736842105 }, "multi_hop": { "correct": 12, "n": 35, "accuracy": 0.34285714285714286 }, "probability": { "correct": 8, "n": 20, "accuracy": 0.4 }, "routing_hard": { "correct": 3, "n": 10, "accuracy": 0.3 }, "temporal_numeric": { "correct": 9, "n": 30, "accuracy": 0.3 }, "tradeoff": { "correct": 6, "n": 12, "accuracy": 0.5 }, "trap": { "correct": 3, "n": 16, "accuracy": 0.1875 } }, "has_distribution": true, "brier_mean": 0.7660523635454545, "ece": 0.20550909090909092, "probability_fidelity": 66.03125, "calibration_score": 62.46471590909091, "onehot": { "ece": 0.6590909090909092, "probability_fidelity": 40.04649999999999, "calibration_score": 20.023249999999994 }, "latency_p50_s": 1.9288412630558014, "latency_p95_s": 2.2888929322361946, "mean_input_tokens": 407.4727272727273, "mean_output_tokens": 0.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 54.35373809070198, "Balanced 33:33:33 (no calibration)": 54.99732499078397, "Emphasis on Accuracy 60:20:20": 47.681275249004166, "Emphasis on Speed 20:60:20": 56.82735959428694, "Emphasis on Cost 20:20:60": 61.393071171072854, "Intelligence only": 38.49083264049513 }, "rank": 33, "rank_under": { "JevBench Score (25:25:25:25)": 33, "Balanced 33:33:33 (no calibration)": 29, "Emphasis on Accuracy 60:20:20": 34, "Emphasis on Speed 20:60:20": 32, "Emphasis on Cost 20:20:60": 12, "Intelligence only": 34 } }, { "key": "open-jev-zefan-2b", "display": "Open-Jev 2B (Zefan Cai)", "class": "jev-rebuild", "open": "yes", "author": "Zefan Cai (@Zefan_Cai)", "repo": "https://github.com/Zefan-Cai/Open-Jev", "licence": "MIT (loader); Apache-2.0 (adapter and pinned Qwen base); CC0-1.0 public training projection", "underlying": "Qwen3.5-2B plus rank-8 LoRA and trained scalar decision head", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our RunPod GPU (H100 80GB HBM3), reached over the internet", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.7916666666666666, "judge": 0.8835616438356164, "hard": 0.42727272727272725 }, "axes": { "intelligence": 60.96359619854646, "calibration": 55.05370286560958, "speed": 73.45427178913715, "cost": 28.123361021000306 }, "jevbench_score": 51.31393833805917, "speed": { "p50_s_raw": 0.664745207875967, "p95_s_raw": 1.450564834475517, "p50_s_adjusted": 1.479490415751934, "p95_s_adjusted": 3.051129668951034, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.6978476755321026, "hard_tier_p95_s": 1.2396420393139123 }, "cost": { "kind": "estimate", "usd_per_1000": 0.24882153558052428, "basis": "ESTIMATE: hosted-provider price, OpenRouter Qwen3.5-9B list price read 2026-09-21 list price $0.1/M in, $0.0/M out (the exact 9B base and a conservative same-family proxy for the unlisted 2B; the decision head generates no output tokens) x 1439 input and 0 output tokens per decision (input tokens measured (the system's own count))", "usd_per_1000_v11_tiers": 0.14393662420382164, "usd_per_1000_hard": 0.39852090909090904, "self_host_sensitivity": null }, "calibration": { "score": 55.05370286560958, "score_label_only_as_onehot": null, "ece_hard": 0.2573327936087847, "probability_fidelity": 61.57396445297609, "brier_hard": 0.7779483924445973, "brier_standard_judge_v11": 0.26468498350008274, "note": null }, "hard": { "run": "runs/open-jev-zefan-2b--hard (job jevbench-zefan-openjev-20260921)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.42727272727272725, "by_family": { "adversarial": { "correct": 6, "n": 12, "accuracy": 0.5 }, "ambiguous": { "correct": 3, "n": 14, "accuracy": 0.21428571428571427 }, "judge_hard": { "correct": 20, "n": 33, "accuracy": 0.6060606060606061 }, "long_policy": { "correct": 8, "n": 38, "accuracy": 0.21052631578947367 }, "multi_hop": { "correct": 18, "n": 35, "accuracy": 0.5142857142857142 }, "probability": { "correct": 7, "n": 20, "accuracy": 0.35 }, "routing_hard": { "correct": 8, "n": 10, "accuracy": 0.8 }, "temporal_numeric": { "correct": 5, "n": 30, "accuracy": 0.16666666666666666 }, "tradeoff": { "correct": 5, "n": 12, "accuracy": 0.4166666666666667 }, "trap": { "correct": 14, "n": 16, "accuracy": 0.875 } }, "has_distribution": true, "brier_mean": 0.7779483924445973, "ece": 0.2573327936087847, "probability_fidelity": 61.57396445297609, "calibration_score": 55.05370286560958, "onehot": { "ece": 0.5727272727272728, "probability_fidelity": 41.585499999999996, "calibration_score": 20.792749999999998 }, "latency_p50_s": 0.6978476755321026, "latency_p95_s": 1.2396420393139123, "mean_input_tokens": 3985.2090909090907, "mean_output_tokens": 0.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 51.31393833805917, "Balanced 33:33:33 (no calibration)": 50.12468069820073, "Emphasis on Accuracy 60:20:20": 54.20747804596064, "Emphasis on Speed 20:60:20": 58.40335479631084, "Emphasis on Cost 20:20:60": 39.7793662889096, "Intelligence only": 60.96359619854648 }, "rank": 34, "rank_under": { "JevBench Score (25:25:25:25)": 34, "Balanced 33:33:33 (no calibration)": 33, "Emphasis on Accuracy 60:20:20": 32, "Emphasis on Speed 20:60:20": 31, "Emphasis on Cost 20:20:60": 37, "Intelligence only": 31 } }, { "key": "opendecision", "display": "OpenDecision (ModernBERT-large zero-shot)", "class": "classifier", "open": "yes", "author": "Deepan Wadhwa", "repo": "https://github.com/deepanwadhwa/OpenDecision", "licence": "Apache-2.0", "underlying": "MoritzLaurer/ModernBERT-large-zeroshot-v2.0 (~400M) used as a zero-shot NLI decision engine", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our RunPod GPU (H100 NVL 96 GB, Canada), reached over the internet from Germany", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 0.875, "standard": 0.625, "judge": 0.7123287671232876, "hard": 0.33181818181818185 }, "axes": { "intelligence": 40.80927227619637, "calibration": 56.092333006283695, "speed": 79.89955851182286, "cost": 75.34414706615897 }, "jevbench_score": 40.58744897671299, "speed": { "p50_s_raw": 0.3378134034574032, "p95_s_raw": 0.5447697393596171, "p50_s_adjusted": 0.8256268069148064, "p95_s_adjusted": 1.2395394787192342, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.35639120638370514, "hard_tier_p95_s": 0.49497928358614446 }, "cost": { "kind": "estimate", "usd_per_1000": 0.006635318352059925, "basis": "ESTIMATE: hosted-provider price, deepinfra encoders of the same size (bge-large, e5-large, Qwen3-Embedding-0.6B) list price $0.01/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 329 input and 0 output tokens per decision (input tokens measured (the system's own count))", "usd_per_1000_v11_tiers": 0.0032921656050955415, "usd_per_1000_hard": 0.011406909090909093, "self_host_sensitivity": null }, "calibration": { "score": 56.092333006283695, "score_label_only_as_onehot": null, "ece_hard": 0.2598902093754573, "probability_fidelity": 64.16270788765885, "brier_hard": 0.8135612674812179, "brier_standard_judge_v11": 0.4328156823860972, "note": null }, "hard": { "run": "runs/opendecision--hard (job jevbench-add-requests-20260919, run 4)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.33181818181818185, "by_family": { "adversarial": { "correct": 10, "n": 12, "accuracy": 0.8333333333333334 }, "ambiguous": { "correct": 2, "n": 14, "accuracy": 0.14285714285714285 }, "judge_hard": { "correct": 16, "n": 33, "accuracy": 0.48484848484848486 }, "long_policy": { "correct": 11, "n": 38, "accuracy": 0.2894736842105263 }, "multi_hop": { "correct": 5, "n": 35, "accuracy": 0.14285714285714285 }, "probability": { "correct": 8, "n": 20, "accuracy": 0.4 }, "routing_hard": { "correct": 1, "n": 10, "accuracy": 0.1 }, "temporal_numeric": { "correct": 7, "n": 30, "accuracy": 0.23333333333333334 }, "tradeoff": { "correct": 3, "n": 12, "accuracy": 0.25 }, "trap": { "correct": 10, "n": 16, "accuracy": 0.625 } }, "has_distribution": true, "brier_mean": 0.8135612674812179, "ece": 0.2598902093754573, "probability_fidelity": 64.16270788765885, "calibration_score": 56.092333006283695, "onehot": { "ece": 0.6681818181818182, "probability_fidelity": 41.01049999999999, "calibration_score": 20.505249999999997 }, "latency_p50_s": 0.35639120638370514, "latency_p95_s": 0.49497928358614446, "mean_input_tokens": 1140.6909090909091, "mean_output_tokens": 0.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 40.58744897671299, "Balanced 33:33:33 (no calibration)": 41.7216948474132, "Emphasis on Accuracy 60:20:20": 35.15214106233547, "Emphasis on Speed 20:60:20": 45.990273687385674, "Emphasis on Cost 20:20:60": 44.92292752979974, "Intelligence only": 27.18545101187708 }, "rank": 35, "rank_under": { "JevBench Score (25:25:25:25)": 35, "Balanced 33:33:33 (no calibration)": 35, "Emphasis on Accuracy 60:20:20": 35, "Emphasis on Speed 20:60:20": 35, "Emphasis on Cost 20:20:60": 28, "Intelligence only": 35 } }, { "key": "openjev-verdict-1.4", "display": "openJev Verdict 1.4", "class": "jev-rebuild", "open": "yes", "author": "Hemant (heman10x)", "repo": "https://huggingface.co/heman10x/rlcd-modernbert-151m", "licence": "Apache-2.0", "underlying": "GLiClass ModernBERT-base fine-tuned decision model, 151M; v1.4 fixed inference engine", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our CPU (4 threads, Ryzen 5 3600)", "endpoint_kind": "cpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 0.8611111111111112, "standard": 0.6770833333333334, "judge": 0.5616438356164384, "hard": 0.37727272727272726 }, "axes": { "intelligence": 38.55664845884808, "calibration": 74.13545833230623, "speed": 78.08595973190066, "cost": 82.35883967864214 }, "jevbench_score": 38.93683519894202, "speed": { "p50_s_raw": 0.31351133808493614, "p95_s_raw": 0.9248626325279473, "p50_s_adjusted": 0.7770226761698723, "p95_s_adjusted": 1.9997252650558945, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads", "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing", "hard_tier_p50_s": 0.6292409487068653, "hard_tier_p95_s": 0.8346716947853564 }, "cost": { "kind": "estimate", "usd_per_1000": 0.003872921348314607, "basis": "ESTIMATE: hosted-provider price, deepinfra base-size encoders (bge-base, e5-base, gte-base, all-mpnet-base) list price $0.005/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 452 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts)", "usd_per_1000_v11_tiers": 0.0022600000000000003, "usd_per_1000_hard": 0.006175, "self_host_sensitivity": null }, "calibration": { "score": 74.13545833230623, "score_label_only_as_onehot": null, "ece_hard": 0.11565818173180892, "probability_fidelity": 71.40255301097424, "brier_hard": 0.6941169946619116, "brier_standard_judge_v11": 0.5357783992553375, "note": null }, "hard": { "run": "runs/openjev-verdict-1.4--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.37727272727272726, "by_family": { "adversarial": { "correct": 8, "n": 12, "accuracy": 0.6666666666666666 }, "ambiguous": { "correct": 6, "n": 14, "accuracy": 0.42857142857142855 }, "judge_hard": { "correct": 18, "n": 33, "accuracy": 0.5454545454545454 }, "long_policy": { "correct": 11, "n": 38, "accuracy": 0.2894736842105263 }, "multi_hop": { "correct": 6, "n": 35, "accuracy": 0.17142857142857143 }, "probability": { "correct": 8, "n": 20, "accuracy": 0.4 }, "routing_hard": { "correct": 3, "n": 10, "accuracy": 0.3 }, "temporal_numeric": { "correct": 9, "n": 30, "accuracy": 0.3 }, "tradeoff": { "correct": 5, "n": 12, "accuracy": 0.4166666666666667 }, "trap": { "correct": 9, "n": 16, "accuracy": 0.5625 } }, "has_distribution": true, "brier_mean": 0.6941169946619116, "ece": 0.11565818173180892, "probability_fidelity": 71.40255301097424, "calibration_score": 74.13545833230623, "onehot": { "ece": 0.6227272727272728, "probability_fidelity": 41.260499999999986, "calibration_score": 20.630249999999993 }, "latency_p50_s": 0.6292409487068653, "latency_p95_s": 0.8346716947853564, "mean_input_tokens": null, "mean_output_tokens": null, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 38.93683519894202, "Balanced 33:33:33 (no calibration)": 37.35820710614536, "Emphasis on Accuracy 60:20:20": 30.730862023210403, "Emphasis on Speed 20:60:20": 40.753434363629175, "Emphasis on Cost 20:20:60": 41.63121828485476, "Intelligence only": 22.927558944480637 }, "rank": 36, "rank_under": { "JevBench Score (25:25:25:25)": 36, "Balanced 33:33:33 (no calibration)": 37, "Emphasis on Accuracy 60:20:20": 38, "Emphasis on Speed 20:60:20": 37, "Emphasis on Cost 20:20:60": 33, "Intelligence only": 38 } }, { "key": "openjev-verdict", "display": "openJev Verdict (heman10x, ModernBERT-base 151M)", "class": "jev-rebuild", "open": "yes", "author": "Hemant (heman10x)", "repo": "https://github.com/Heman10x-NGU/openJev-verdict-2.0", "licence": "Apache-2.0", "underlying": "GLiClass ModernBERT-base (knowledgator/gliclass-modern-base-v2.0) fine-tuned, 151M", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our CPU (4 threads, Ryzen 5 3600)", "endpoint_kind": "cpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 0.8611111111111112, "standard": 0.65625, "judge": 0.6095890410958904, "hard": 0.38181818181818183 }, "axes": { "intelligence": 39.80526702710233, "calibration": 51.29640397786167, "speed": 76.68207818819491, "cost": 83.05556459946699 }, "jevbench_score": 38.059545049974176, "speed": { "p50_s_raw": 0.27805980294942856, "p95_s_raw": 1.4451411496847864, "p50_s_adjusted": 0.7061196058988571, "p95_s_adjusted": 3.0402822993695726, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads", "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing", "hard_tier_p50_s": 0.750414352864027, "hard_tier_p95_s": 1.7721023652702563 }, "cost": { "kind": "estimate", "usd_per_1000": 0.00367125468164794, "basis": "ESTIMATE: hosted-provider price, deepinfra base-size encoders (bge-base, e5-base, gte-base, all-mpnet-base) list price $0.005/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 383 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json]", "usd_per_1000_v11_tiers": 0.0019170382165605096, "usd_per_1000_hard": 0.006175, "self_host_sensitivity": null }, "calibration": { "score": 51.29640397786167, "score_label_only_as_onehot": null, "ece_hard": 0.30116600532276583, "probability_fidelity": 62.826009020276516, "brier_hard": 0.8456557054244584, "brier_standard_judge_v11": 0.4858771320072627, "note": null }, "hard": { "run": "runs/openjev-verdict--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.38181818181818183, "by_family": { "adversarial": { "correct": 8, "n": 12, "accuracy": 0.6666666666666666 }, "ambiguous": { "correct": 6, "n": 14, "accuracy": 0.42857142857142855 }, "judge_hard": { "correct": 18, "n": 33, "accuracy": 0.5454545454545454 }, "long_policy": { "correct": 11, "n": 38, "accuracy": 0.2894736842105263 }, "multi_hop": { "correct": 5, "n": 35, "accuracy": 0.14285714285714285 }, "probability": { "correct": 9, "n": 20, "accuracy": 0.45 }, "routing_hard": { "correct": 3, "n": 10, "accuracy": 0.3 }, "temporal_numeric": { "correct": 12, "n": 30, "accuracy": 0.4 }, "tradeoff": { "correct": 4, "n": 12, "accuracy": 0.3333333333333333 }, "trap": { "correct": 8, "n": 16, "accuracy": 0.5 } }, "has_distribution": true, "brier_mean": 0.8456557054244584, "ece": 0.30116600532276583, "probability_fidelity": 62.826009020276516, "calibration_score": 51.29640397786167, "onehot": { "ece": 0.6181818181818182, "probability_fidelity": 43.343999999999994, "calibration_score": 21.671999999999997 }, "latency_p50_s": 0.750414352864027, "latency_p95_s": 1.7721023652702563, "mean_input_tokens": null, "mean_output_tokens": null, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 38.059545049974176, "Balanced 33:33:33 (no calibration)": 40.11210544380774, "Emphasis on Accuracy 60:20:20": 33.3209919353683, "Emphasis on Speed 20:60:20": 43.31309864675517, "Emphasis on Cost 20:20:60": 44.718702975382016, "Intelligence only": 25.227929942929453 }, "rank": 37, "rank_under": { "JevBench Score (25:25:25:25)": 37, "Balanced 33:33:33 (no calibration)": 36, "Emphasis on Accuracy 60:20:20": 36, "Emphasis on Speed 20:60:20": 36, "Emphasis on Cost 20:20:60": 29, "Intelligence only": 37 } }, { "key": "kev-0.5b", "display": "kev 0.5B", "class": "jev-rebuild", "open": true, "author": "Jared Palmer", "repo": "https://github.com/jaredpalmer/kev", "licence": "Apache-2.0", "underlying": "Qwen2.5-0.5B + LoRA + learned pointer head; jaredpalmer/kev-0.5b", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our RunPod GPU (GeForce RTX 3090 24 GB, community cloud CA), reached over the internet", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 0.9583333333333334, "standard": 0.5208333333333334, "judge": 0.7123287671232876, "hard": 0.3090909090909091 }, "axes": { "intelligence": 38.170465529254024, "calibration": 47.367924010582875, "speed": 76.95661397000733, "cost": 76.08687775906971 }, "jevbench_score": 33.24350332804402, "speed": { "p50_s_raw": 0.43036095052957535, "p95_s_raw": 0.9219581566751004, "p50_s_adjusted": 1.0107219010591506, "p95_s_adjusted": 1.9939163133502007, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.5884033516049385, "hard_tier_p95_s": 1.2630689594894642 }, "cost": { "kind": "estimate", "usd_per_1000": 0.0062676404494382025, "basis": "ESTIMATE: hosted-provider price, DeepInfra Qwen3-Embedding-0.6B size-class reference list price $0.01/M in, $0.0/M out (a <=0.6B one-pass model with no generated output) x 279 input and 0 output tokens per decision (input tokens measured (the system's own count))", "usd_per_1000_v11_tiers": 0.0027941401273885347, "usd_per_1000_hard": 0.011225272727272728, "self_host_sensitivity": null }, "calibration": { "score": 47.367924010582875, "score_label_only_as_onehot": null, "ece_hard": 0.3138481166298448, "probability_fidelity": 57.50547134713471, "brier_hard": 0.8464522868993845, "brier_standard_judge_v11": 0.46104057750389293, "note": null }, "hard": { "run": "runs/kev-0.5b--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.3090909090909091, "by_family": { "adversarial": { "correct": 4, "n": 12, "accuracy": 0.3333333333333333 }, "ambiguous": { "correct": 4, "n": 14, "accuracy": 0.2857142857142857 }, "judge_hard": { "correct": 20, "n": 33, "accuracy": 0.6060606060606061 }, "long_policy": { "correct": 7, "n": 38, "accuracy": 0.18421052631578946 }, "multi_hop": { "correct": 11, "n": 35, "accuracy": 0.3142857142857143 }, "probability": { "correct": 4, "n": 20, "accuracy": 0.2 }, "routing_hard": { "correct": 0, "n": 10, "accuracy": 0.0 }, "temporal_numeric": { "correct": 4, "n": 30, "accuracy": 0.13333333333333333 }, "tradeoff": { "correct": 5, "n": 12, "accuracy": 0.4166666666666667 }, "trap": { "correct": 9, "n": 16, "accuracy": 0.5625 } }, "has_distribution": true, "brier_mean": 0.8464522868993845, "ece": 0.3138481166298448, "probability_fidelity": 57.50547134713471, "calibration_score": 47.367924010582875, "onehot": { "ece": 0.6909090909090909, "probability_fidelity": 30.452500000000004, "calibration_score": 15.226250000000002 }, "latency_p50_s": 0.5884033516049385, "latency_p95_s": 1.2630689594894642, "mean_input_tokens": 1122.5272727272727, "mean_output_tokens": 62.163636363636364, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 33.24350332804402, "Balanced 33:33:33 (no calibration)": 35.36790170106862, "Emphasis on Accuracy 60:20:20": 29.38072589185842, "Emphasis on Speed 20:60:20": 38.89286347386012, "Emphasis on Cost 20:20:60": 38.716442423141594, "Intelligence only": 22.245509717928044 }, "rank": 38, "rank_under": { "JevBench Score (25:25:25:25)": 38, "Balanced 33:33:33 (no calibration)": 39, "Emphasis on Accuracy 60:20:20": 39, "Emphasis on Speed 20:60:20": 38, "Emphasis on Cost 20:20:60": 38, "Intelligence only": 39 } }, { "key": "gliner2-large", "display": "GLiNER2 large (Fastino)", "class": "classifier", "open": "yes", "author": "Fastino", "repo": "https://huggingface.co/fastino/gliner2-large-v1", "licence": "Apache-2.0", "underlying": "GLiNER2 large schema-conditioned extractor (DeBERTa-v3-large encoder)", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our CPU (4 threads, Ryzen 5 3600)", "endpoint_kind": "cpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 0.9861111111111112, "standard": 0.625, "judge": 0.6095890410958904, "hard": 0.36363636363636365 }, "axes": { "intelligence": 40.14699648103604, "calibration": 24.31885823276304, "speed": 61.65849142544844, "cost": 73.3279398087227 }, "jevbench_score": 29.55158409895235, "speed": { "p50_s_raw": 1.0967330671846867, "p95_s_raw": 14.48837994225323, "p50_s_adjusted": 2.3434661343693732, "p95_s_adjusted": 29.12675988450646, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads", "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing", "hard_tier_p50_s": 2.5048606134951115, "hard_tier_p95_s": 64.29295974373817 }, "cost": { "kind": "estimate", "usd_per_1000": 0.007745842696629214, "basis": "ESTIMATE: hosted-provider price, deepinfra encoders of the same size (bge-large, e5-large, Qwen3-Embedding-0.6B) list price $0.01/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 452 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts)", "usd_per_1000_v11_tiers": 0.004520000000000001, "usd_per_1000_hard": 0.01235, "self_host_sensitivity": null }, "calibration": { "score": 24.31885823276304, "score_label_only_as_onehot": null, "ece_hard": 0.4661084924231875, "probability_fidelity": 41.85941495016359, "brier_hard": 1.0843070639080452, "brier_standard_judge_v11": 0.6502070879795632, "note": null }, "hard": { "run": "runs/gliner2-large--hard (job jevbench-add-requests-20260919, run 4)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.36363636363636365, "by_family": { "adversarial": { "correct": 7, "n": 12, "accuracy": 0.5833333333333334 }, "ambiguous": { "correct": 4, "n": 14, "accuracy": 0.2857142857142857 }, "judge_hard": { "correct": 18, "n": 33, "accuracy": 0.5454545454545454 }, "long_policy": { "correct": 10, "n": 38, "accuracy": 0.2631578947368421 }, "multi_hop": { "correct": 9, "n": 35, "accuracy": 0.2571428571428571 }, "probability": { "correct": 5, "n": 20, "accuracy": 0.25 }, "routing_hard": { "correct": 7, "n": 10, "accuracy": 0.7 }, "temporal_numeric": { "correct": 7, "n": 30, "accuracy": 0.23333333333333334 }, "tradeoff": { "correct": 2, "n": 12, "accuracy": 0.16666666666666666 }, "trap": { "correct": 11, "n": 16, "accuracy": 0.6875 } }, "has_distribution": true, "brier_mean": 1.0843070639080452, "ece": 0.4661084924231875, "probability_fidelity": 41.85941495016359, "calibration_score": 24.31885823276304, "onehot": { "ece": 0.6363636363636364, "probability_fidelity": 33.821, "calibration_score": 16.9105 }, "latency_p50_s": 2.5048606134951115, "latency_p95_s": 64.29295974373817, "mean_input_tokens": null, "mean_output_tokens": null, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 29.55158409895235, "Balanced 33:33:33 (no calibration)": 36.50378719025428, "Emphasis on Accuracy 60:20:20": 31.813424670173813, "Emphasis on Speed 20:60:20": 37.76994537761949, "Emphasis on Cost 20:20:60": 40.48153688756807, "Intelligence only": 25.883271696448123 }, "rank": 39, "rank_under": { "JevBench Score (25:25:25:25)": 39, "Balanced 33:33:33 (no calibration)": 38, "Emphasis on Accuracy 60:20:20": 37, "Emphasis on Speed 20:60:20": 39, "Emphasis on Cost 20:20:60": 36, "Intelligence only": 36 } }, { "key": "smalljev", "display": "smalljev semantic-v9", "class": "jev-rebuild", "open": "yes", "author": "Aditya (isHeSatoshi)", "repo": "https://github.com/isHeSatoshi/smalljev", "licence": "Apache-2.0", "underlying": "MiniCPM5-2B-Base, 2.5B dense, with LoRA and native semantic decision heads", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our GPU (lium.io A6000 48 GB), reached over the internet from Germany; serial, one request at a time", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 0.9722222222222222, "standard": 0.6875, "judge": 0.4041095890410959, "hard": 0.38181818181818183 }, "axes": { "intelligence": 35.12937048856184, "calibration": 58.88025149088166, "speed": 79.81829102912181, "cost": 57.872598102653555 }, "jevbench_score": 27.444453253651357, "speed": { "p50_s_raw": 0.4138255603611469, "p95_s_raw": 0.45828209072351456, "p50_s_adjusted": 0.9776511207222939, "p95_s_adjusted": 1.066564181447029, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.42050960287451744, "hard_tier_p95_s": 0.8517502576112745 }, "cost": { "kind": "estimate", "usd_per_1000": 0.025365692883895133, "basis": "ESTIMATE: hosted-provider price, submitted Qwen/Qwen2.5-3B-Instruct hosted reference list price $0.04/M in, $0.0/M out (the author's documented reference for the same approximate size class; one forward pass, nothing generated) x 329 input and 0 output tokens per decision (input tokens measured (the system's own count))", "usd_per_1000_v11_tiers": 0.013142292993630575, "usd_per_1000_hard": 0.04281181818181819, "self_host_sensitivity": null }, "calibration": { "score": 58.88025149088166, "score_label_only_as_onehot": null, "ece_hard": 0.2262606866657734, "probability_fidelity": 63.012640314918, "brier_hard": 0.7819475766009777, "brier_standard_judge_v11": 0.5889297804477662, "note": null }, "hard": { "run": "runs/smalljev--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.38181818181818183, "by_family": { "adversarial": { "correct": 7, "n": 12, "accuracy": 0.5833333333333334 }, "ambiguous": { "correct": 2, "n": 14, "accuracy": 0.14285714285714285 }, "judge_hard": { "correct": 19, "n": 33, "accuracy": 0.5757575757575758 }, "long_policy": { "correct": 9, "n": 38, "accuracy": 0.23684210526315788 }, "multi_hop": { "correct": 7, "n": 35, "accuracy": 0.2 }, "probability": { "correct": 9, "n": 20, "accuracy": 0.45 }, "routing_hard": { "correct": 4, "n": 10, "accuracy": 0.4 }, "temporal_numeric": { "correct": 11, "n": 30, "accuracy": 0.36666666666666664 }, "tradeoff": { "correct": 6, "n": 12, "accuracy": 0.5 }, "trap": { "correct": 10, "n": 16, "accuracy": 0.625 } }, "has_distribution": true, "brier_mean": 0.7819475766009777, "ece": 0.2262606866657734, "probability_fidelity": 63.012640314918, "calibration_score": 58.88025149088166, "onehot": { "ece": 0.6181818181818182, "probability_fidelity": 39.76250000000001, "calibration_score": 19.881250000000005 }, "latency_p50_s": 0.42050960287451744, "latency_p95_s": 0.8517502576112745, "mean_input_tokens": 1070.2954545454545, "mean_output_tokens": 0.0, "charged_usd": 0.009418599999999996 }, "presets": { "JevBench Score (25:25:25:25)": 27.444453253651357, "Balanced 33:33:33 (no calibration)": 26.924603522460426, "Emphasis on Accuracy 60:20:20": 22.57969272223544, "Emphasis on Speed 20:60:20": 31.353848139077847, "Emphasis on Cost 20:20:60": 27.57014641272405, "Intelligence only": 17.34087842666018 }, "rank": 40, "rank_under": { "JevBench Score (25:25:25:25)": 40, "Balanced 33:33:33 (no calibration)": 41, "Emphasis on Accuracy 60:20:20": 41, "Emphasis on Speed 20:60:20": 41, "Emphasis on Cost 20:20:60": 41, "Intelligence only": 41 } }, { "key": "gliner2", "display": "GLiNER2 (Fastino, gliner2.5-base)", "class": "classifier", "open": "yes", "author": "Fastino", "repo": "https://github.com/fastino-ai/GLiNER2", "licence": "Apache-2.0", "underlying": "DeBERTa-v3-base schema-conditioned extractor (GLiNER2.5 boundary architecture), 194M", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our CPU (4 threads, Ryzen 5 3600)", "endpoint_kind": "cpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 0.9722222222222222, "standard": 0.6666666666666666, "judge": 0.4589041095890411, "hard": 0.36363636363636365 }, "axes": { "intelligence": 35.62144224229537, "calibration": 23.668537162731784, "speed": 71.82859826025441, "cost": 83.05556459946699 }, "jevbench_score": 24.036445808742133, "speed": { "p50_s_raw": 0.31302378326654434, "p95_s_raw": 4.153845678269863, "p50_s_adjusted": 0.7760475665330887, "p95_s_adjusted": 8.457691356539726, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads", "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing", "hard_tier_p50_s": 0.9508068449795246, "hard_tier_p95_s": 24.590944871306384 }, "cost": { "kind": "estimate", "usd_per_1000": 0.00367125468164794, "basis": "ESTIMATE: hosted-provider price, deepinfra base-size encoders (bge-base, e5-base, gte-base, all-mpnet-base) list price $0.005/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 383 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json]", "usd_per_1000_v11_tiers": 0.0019170382165605096, "usd_per_1000_hard": 0.006175, "self_host_sensitivity": null }, "calibration": { "score": 23.668537162731784, "score_label_only_as_onehot": null, "ece_hard": 0.47177480337294664, "probability_fidelity": 41.6920350000529, "brier_hard": 1.0505351022898706, "brier_standard_judge_v11": 0.8269065970344461, "note": null }, "hard": { "run": "runs/gliner2--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.36363636363636365, "by_family": { "adversarial": { "correct": 5, "n": 12, "accuracy": 0.4166666666666667 }, "ambiguous": { "correct": 5, "n": 14, "accuracy": 0.35714285714285715 }, "judge_hard": { "correct": 15, "n": 33, "accuracy": 0.45454545454545453 }, "long_policy": { "correct": 10, "n": 38, "accuracy": 0.2631578947368421 }, "multi_hop": { "correct": 13, "n": 35, "accuracy": 0.37142857142857144 }, "probability": { "correct": 6, "n": 20, "accuracy": 0.3 }, "routing_hard": { "correct": 7, "n": 10, "accuracy": 0.7 }, "temporal_numeric": { "correct": 6, "n": 30, "accuracy": 0.2 }, "tradeoff": { "correct": 6, "n": 12, "accuracy": 0.5 }, "trap": { "correct": 7, "n": 16, "accuracy": 0.4375 } }, "has_distribution": true, "brier_mean": 1.0505351022898706, "ece": 0.47177480337294664, "probability_fidelity": 41.6920350000529, "calibration_score": 23.668537162731784, "onehot": { "ece": 0.6363636363636364, "probability_fidelity": 33.6155, "calibration_score": 16.80775 }, "latency_p50_s": 0.9508068449795246, "latency_p95_s": 24.590944871306384, "mean_input_tokens": null, "mean_output_tokens": null, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 24.036445808742133, "Balanced 33:33:33 (no calibration)": 30.288344165438957, "Emphasis on Accuracy 60:20:20": 24.640136790659795, "Emphasis on Speed 20:60:20": 32.61951288341773, "Emphasis on Cost 20:20:60": 34.57052769175062, "Intelligence only": 18.07983609354188 }, "rank": 41, "rank_under": { "JevBench Score (25:25:25:25)": 41, "Balanced 33:33:33 (no calibration)": 40, "Emphasis on Accuracy 60:20:20": 40, "Emphasis on Speed 20:60:20": 40, "Emphasis on Cost 20:20:60": 39, "Intelligence only": 40 } }, { "key": "open-jev-deberta-v3-large", "display": "open-jev-deberta-v3-large (local CPU)", "class": "jev-rebuild", "open": "yes", "author": "Kotoba Labs", "repo": "https://github.com/kotoba-lang/typed-decisions", "licence": "Apache-2.0 (model card); DeBERTa-v3 keeps its own terms", "underlying": "microsoft/deberta-v3-large", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our CPU (2 threads, Ryzen 5 3600)", "endpoint_kind": "cpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 1.0, "standard": 0.4895833333333333, "judge": 0.5342465753424658, "hard": 0.36363636363636365 }, "axes": { "intelligence": 31.88915182127158, "calibration": 66.3822600310561, "speed": 65.97921316871283, "cost": 74.02828721517697 }, "jevbench_score": 23.065925583996304, "speed": { "p50_s_raw": 1.767673410475254, "p95_s_raw": 3.349288306012749, "p50_s_adjusted": 3.685346820950508, "p95_s_adjusted": 6.848576612025498, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "local, 2 CPU threads of a Ryzen 5 3600", "hard_tier_p50_s": 2.63558766245842, "hard_tier_p95_s": 4.7398640830069665 }, "cost": { "kind": "estimate", "usd_per_1000": 0.007340468164794008, "basis": "ESTIMATE: hosted-provider price, deepinfra encoders of the same size (bge-large, e5-large, Qwen3-Embedding-0.6B) list price $0.01/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 383 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | ESTIMATE: deepinfra encoders of the same size (bge-large, e5-large, Qwen3-Embedding-0.6B) $0.01/M in, $0.0/M out x 1235 in / 0 out tokens per hard decision", "usd_per_1000_v11_tiers": 0.003834076433121019, "usd_per_1000_hard": 0.012345045454545454, "self_host_sensitivity": { "usd_per_1000": 0.014168273865544099, "score": 71.21707642612242, "machine": "Hetzner CX22 (2 vCPU)", "usd_per_h": 0.0072, "concurrency": 1, "utilisation": 0.3, "p50_s_used": 2.1252410798316146, "decisions_per_hour": 508.1776417033919 } }, "calibration": { "score": 66.3822600310561, "score_label_only_as_onehot": null, "ece_hard": 0.17293095619443358, "probability_fidelity": 67.35071130099892, "brier_hard": 0.7102063910056119, "brier_standard_judge_v11": 0.6512140520637422, "note": null }, "hard": { "run": "runs/open-jev-deberta-v3-large--hard", "n_items": 220, "n_ok": 218, "n_attempted": 220, "coverage": 1.0, "success_rate": 0.990909090909091, "accuracy": 0.36363636363636365, "by_family": { "adversarial": { "correct": 8, "n": 12, "accuracy": 0.6666666666666666 }, "ambiguous": { "correct": 6, "n": 14, "accuracy": 0.42857142857142855 }, "judge_hard": { "correct": 18, "n": 33, "accuracy": 0.5454545454545454 }, "long_policy": { "correct": 13, "n": 38, "accuracy": 0.34210526315789475 }, "multi_hop": { "correct": 13, "n": 35, "accuracy": 0.37142857142857144 }, "probability": { "correct": 6, "n": 20, "accuracy": 0.3 }, "routing_hard": { "correct": 2, "n": 10, "accuracy": 0.2 }, "temporal_numeric": { "correct": 4, "n": 30, "accuracy": 0.13333333333333333 }, "tradeoff": { "correct": 5, "n": 12, "accuracy": 0.4166666666666667 }, "trap": { "correct": 5, "n": 16, "accuracy": 0.3125 } }, "has_distribution": true, "brier_mean": 0.7102063910056119, "ece": 0.17293095619443358, "probability_fidelity": 67.35071130099892, "calibration_score": 66.3822600310561, "onehot": { "ece": 0.6330275229357798, "probability_fidelity": 35.785999999999994, "calibration_score": 17.892999999999997 }, "latency_p50_s": 2.63558766245842, "latency_p95_s": 4.7398640830069665, "mean_input_tokens": null, "mean_output_tokens": null, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 23.065925583996304, "Balanced 33:33:33 (no calibration)": 21.885771542636757, "Emphasis on Accuracy 60:20:20": 17.753855270934945, "Emphasis on Speed 20:60:20": 23.746430712788253, "Emphasis on Cost 20:20:60": 24.865349529792606, "Intelligence only": 12.971461046206885 }, "rank": 42, "rank_under": { "JevBench Score (25:25:25:25)": 42, "Balanced 33:33:33 (no calibration)": 42, "Emphasis on Accuracy 60:20:20": 42, "Emphasis on Speed 20:60:20": 42, "Emphasis on Cost 20:20:60": 42, "Intelligence only": 42 } }, { "key": "gliner2.5-multi", "display": "GLiNER2.5 multi (Fastino, 287M)", "class": "classifier", "open": "yes", "author": "Fastino", "repo": "https://huggingface.co/fastino/gliner2.5-multi-v1", "licence": "Apache-2.0", "underlying": "GLiNER2.5 multilingual schema-conditioned extractor, 287M", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our CPU (4 threads, Ryzen 5 3600)", "endpoint_kind": "cpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 0.9027777777777778, "standard": 0.5104166666666666, "judge": 0.4383561643835616, "hard": 0.37727272727272726 }, "axes": { "intelligence": 27.664703779318668, "calibration": 56.141750517866484, "speed": 67.79966551109104, "cost": 82.35883967864214 }, "jevbench_score": 16.61305398280731, "speed": { "p50_s_raw": 0.42792757973074913, "p95_s_raw": 8.17526703067124, "p50_s_adjusted": 1.0058551594614982, "p95_s_adjusted": 16.500534061342478, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads", "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing", "hard_tier_p50_s": 1.2314840070903301, "hard_tier_p95_s": 41.43838266544044 }, "cost": { "kind": "estimate", "usd_per_1000": 0.003872921348314607, "basis": "ESTIMATE: hosted-provider price, deepinfra base-size encoders (bge-base, e5-base, gte-base, all-mpnet-base) list price $0.005/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 452 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts)", "usd_per_1000_v11_tiers": 0.0022600000000000003, "usd_per_1000_hard": 0.006175, "self_host_sensitivity": null }, "calibration": { "score": 56.141750517866484, "score_label_only_as_onehot": null, "ece_hard": 0.2653272879394618, "probability_fidelity": 65.34895862362534, "brier_hard": 0.8370279252684946, "brier_standard_judge_v11": 0.769624024073474, "note": null }, "hard": { "run": "runs/gliner2.5-multi--hard (job jevbench-add-requests-20260919, run 3)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.37727272727272726, "by_family": { "adversarial": { "correct": 5, "n": 12, "accuracy": 0.4166666666666667 }, "ambiguous": { "correct": 6, "n": 14, "accuracy": 0.42857142857142855 }, "judge_hard": { "correct": 17, "n": 33, "accuracy": 0.5151515151515151 }, "long_policy": { "correct": 14, "n": 38, "accuracy": 0.3684210526315789 }, "multi_hop": { "correct": 13, "n": 35, "accuracy": 0.37142857142857144 }, "probability": { "correct": 9, "n": 20, "accuracy": 0.45 }, "routing_hard": { "correct": 1, "n": 10, "accuracy": 0.1 }, "temporal_numeric": { "correct": 9, "n": 30, "accuracy": 0.3 }, "tradeoff": { "correct": 4, "n": 12, "accuracy": 0.3333333333333333 }, "trap": { "correct": 5, "n": 16, "accuracy": 0.3125 } }, "has_distribution": true, "brier_mean": 0.8370279252684946, "ece": 0.2653272879394618, "probability_fidelity": 65.34895862362534, "calibration_score": 56.141750517866484, "onehot": { "ece": 0.6227272727272728, "probability_fidelity": 41.69349999999999, "calibration_score": 20.846749999999997 }, "latency_p50_s": 1.2314840070903301, "latency_p95_s": 41.43838266544044, "mean_input_tokens": null, "mean_output_tokens": null, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 16.61305398280731, "Balanced 33:33:33 (no calibration)": 16.42605622482007, "Emphasis on Accuracy 60:20:20": 12.602456550071672, "Emphasis on Speed 20:60:20": 18.037478214016055, "Emphasis on Cost 20:20:60": 19.497049163221, "Intelligence only": 8.469115668973945 }, "rank": 43, "rank_under": { "JevBench Score (25:25:25:25)": 43, "Balanced 33:33:33 (no calibration)": 43, "Emphasis on Accuracy 60:20:20": 43, "Emphasis on Speed 20:60:20": 43, "Emphasis on Cost 20:20:60": 43, "Intelligence only": 43 } }, { "key": "gliner2.5-small", "display": "GLiNER2.5 small (Fastino, 74M)", "class": "classifier", "open": "yes", "author": "Fastino", "repo": "https://huggingface.co/fastino/gliner2.5-small-v1", "licence": "Apache-2.0", "underlying": "GLiNER2.5 small schema-conditioned extractor, 74M", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our CPU (4 threads, Ryzen 5 3600)", "endpoint_kind": "cpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 0.8333333333333334, "standard": 0.4791666666666667, "judge": 0.5, "hard": 0.33181818181818185 }, "axes": { "intelligence": 25.618919552106576, "calibration": 47.1625470589488, "speed": 77.83454067554766, "cost": 82.35883967864214 }, "jevbench_score": 13.84974486725046, "speed": { "p50_s_raw": 0.11413825303316116, "p95_s_raw": 2.1012388937175266, "p50_s_adjusted": 0.37827650606632235, "p95_s_adjusted": 4.3524777874350535, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads", "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing", "hard_tier_p50_s": 0.30439889430999756, "hard_tier_p95_s": 11.560181383416035 }, "cost": { "kind": "estimate", "usd_per_1000": 0.003872921348314607, "basis": "ESTIMATE: hosted-provider price, deepinfra base-size encoders (bge-base, e5-base, gte-base, all-mpnet-base) list price $0.005/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 452 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts)", "usd_per_1000_v11_tiers": 0.0022600000000000003, "usd_per_1000_hard": 0.006175, "self_host_sensitivity": null }, "calibration": { "score": 47.1625470589488, "score_label_only_as_onehot": null, "ece_hard": 0.3264417127452114, "probability_fidelity": 59.613436666939876, "brier_hard": 0.9006336546139394, "brier_standard_judge_v11": 0.7413305857792032, "note": null }, "hard": { "run": "runs/gliner2.5-small--hard (job jevbench-add-requests-20260919, run 3)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.33181818181818185, "by_family": { "adversarial": { "correct": 5, "n": 12, "accuracy": 0.4166666666666667 }, "ambiguous": { "correct": 2, "n": 14, "accuracy": 0.14285714285714285 }, "judge_hard": { "correct": 18, "n": 33, "accuracy": 0.5454545454545454 }, "long_policy": { "correct": 12, "n": 38, "accuracy": 0.3157894736842105 }, "multi_hop": { "correct": 10, "n": 35, "accuracy": 0.2857142857142857 }, "probability": { "correct": 7, "n": 20, "accuracy": 0.35 }, "routing_hard": { "correct": 2, "n": 10, "accuracy": 0.2 }, "temporal_numeric": { "correct": 6, "n": 30, "accuracy": 0.2 }, "tradeoff": { "correct": 7, "n": 12, "accuracy": 0.5833333333333334 }, "trap": { "correct": 4, "n": 16, "accuracy": 0.25 } }, "has_distribution": true, "brier_mean": 0.9006336546139394, "ece": 0.3264417127452114, "probability_fidelity": 59.613436666939876, "calibration_score": 47.1625470589488, "onehot": { "ece": 0.6681818181818182, "probability_fidelity": 37.12949999999999, "calibration_score": 18.564749999999997 }, "latency_p50_s": 0.30439889430999756, "latency_p95_s": 11.560181383416035, "mean_input_tokens": null, "mean_output_tokens": null, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 13.84974486725046, "Balanced 33:33:33 (no calibration)": 14.376816951696135, "Emphasis on Accuracy 60:20:20": 10.609492604474516, "Emphasis on Speed 20:60:20": 16.547760902469822, "Emphasis on Cost 20:20:60": 16.926001623134436, "Intelligence only": 6.725776340118341 }, "rank": 44, "rank_under": { "JevBench Score (25:25:25:25)": 44, "Balanced 33:33:33 (no calibration)": 44, "Emphasis on Accuracy 60:20:20": 44, "Emphasis on Speed 20:60:20": 44, "Emphasis on Cost 20:20:60": 44, "Intelligence only": 44 } }, { "key": "mxbai-rerank-base-v2", "display": "Mixedbread mxbai-rerank-base-v2", "class": "reranker", "open": "yes", "author": "Mixedbread", "repo": "https://huggingface.co/mixedbread-ai/mxbai-rerank-base-v2", "licence": "Apache-2.0", "underlying": "494M cross-encoder", "has_distribution": true, "probability_source": [ "public_calibrated_reranker_softmax" ], "endpoint_condition": "our GPU (lium.io A6000 48 GB), serial, one option batch per decision", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 0.4444444444444444, "standard": 0.3333333333333333, "judge": 0.2671232876712329, "hard": 0.4 }, "axes": { "intelligence": 6.693302965594308, "calibration": 83.14828824044193, "speed": 87.50360537571527, "cost": 67.92989463663068 }, "jevbench_score": 0.764251199794221, "speed": { "p50_s_raw": 0.06881788885220885, "p95_s_raw": 0.23386348099447787, "p50_s_adjusted": 0.2876357777044177, "p95_s_adjusted": 0.6177269619889557, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.07308950461447239, "hard_tier_p95_s": 0.31191255310550325 }, "cost": { "kind": "measured", "usd_per_1000": 0.011722048450570633, "basis": "MEASURED model inference time x lium.io A6000 tariff USD 0.42/hour", "usd_per_1000_v11_tiers": 0.009748850236231591, "usd_per_1000_hard": 0.014538340447399992, "self_host_sensitivity": "Whole five-model rental including setup/download was USD 0.21." }, "calibration": { "score": 83.14828824044193, "score_label_only_as_onehot": null, "ece_hard": 0.02692298605433799, "probability_fidelity": 71.68117369175145, "brier_hard": 0.6540462249946867, "brier_standard_judge_v11": 0.6896480856085945, "note": null }, "hard": { "run": "runs/mxbai-rerank-base-v2--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.4, "by_family": { "adversarial": { "correct": 6, "n": 12, "accuracy": 0.5 }, "ambiguous": { "correct": 4, "n": 14, "accuracy": 0.2857142857142857 }, "judge_hard": { "correct": 18, "n": 33, "accuracy": 0.5454545454545454 }, "long_policy": { "correct": 14, "n": 38, "accuracy": 0.3684210526315789 }, "multi_hop": { "correct": 16, "n": 35, "accuracy": 0.45714285714285713 }, "probability": { "correct": 7, "n": 20, "accuracy": 0.35 }, "routing_hard": { "correct": 2, "n": 10, "accuracy": 0.2 }, "temporal_numeric": { "correct": 7, "n": 30, "accuracy": 0.23333333333333334 }, "tradeoff": { "correct": 7, "n": 12, "accuracy": 0.5833333333333334 }, "trap": { "correct": 7, "n": 16, "accuracy": 0.4375 } }, "has_distribution": true, "brier_mean": 0.6540462249946867, "ece": 0.02692298605433799, "probability_fidelity": 71.68117369175145, "calibration_score": 83.14828824044193, "onehot": { "ece": 0.6, "probability_fidelity": 33.347, "calibration_score": 16.6735 }, "latency_p50_s": 0.07308950461447239, "latency_p95_s": 0.31191255310550325, "mean_input_tokens": 4350.459090909091, "mean_output_tokens": 0.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 0.764251199794221, "Balanced 33:33:33 (no calibration)": 0.6117632935306742, "Emphasis on Accuracy 60:20:20": 0.31881784779847766, "Emphasis on Speed 20:60:20": 0.8914496347285552, "Emphasis on Cost 20:20:60": 0.8055839570069108, "Intelligence only": 0.11994480462665809 }, "rank": 45, "rank_under": { "JevBench Score (25:25:25:25)": 45, "Balanced 33:33:33 (no calibration)": 45, "Emphasis on Accuracy 60:20:20": 45, "Emphasis on Speed 20:60:20": 45, "Emphasis on Cost 20:20:60": 45, "Intelligence only": 45 } }, { "key": "bge-reranker-v2-m3", "display": "BAAI bge-reranker-v2-m3", "class": "reranker", "open": "yes", "author": "BAAI", "repo": "https://huggingface.co/BAAI/bge-reranker-v2-m3", "licence": "Apache-2.0", "underlying": "568M multilingual cross-encoder", "has_distribution": true, "probability_source": [ "public_calibrated_reranker_softmax" ], "endpoint_condition": "our GPU (lium.io A6000 48 GB), serial, one option batch per decision", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 0.4305555555555556, "standard": 0.3645833333333333, "judge": 0.08904109589041095, "hard": 0.36818181818181817 }, "axes": { "intelligence": 6.263689402102906, "calibration": 83.82649106525265, "speed": 89.52513700533652, "cost": 73.3733106383488 }, "jevbench_score": 0.6763072957019215, "speed": { "p50_s_raw": 0.034560746513307095, "p95_s_raw": 0.17954895906150325, "p50_s_adjusted": 0.21912149302661418, "p95_s_adjusted": 0.5090979181230065, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.040032369550317526, "hard_tier_p95_s": 0.24636488300748166 }, "cost": { "kind": "measured", "usd_per_1000": 0.007718915951068509, "basis": "MEASURED model inference time x lium.io A6000 tariff USD 0.42/hour", "usd_per_1000_v11_tiers": 0.005582540915812308, "usd_per_1000_hard": 0.010768105774115997, "self_host_sensitivity": "Whole five-model rental including setup/download was USD 0.21." }, "calibration": { "score": 83.82649106525265, "score_label_only_as_onehot": null, "ece_hard": 0.022679951965887575, "probability_fidelity": 72.18897252368282, "brier_hard": 0.6627383026655216, "brier_standard_judge_v11": 0.7142212126429204, "note": null }, "hard": { "run": "runs/bge-reranker-v2-m3--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.36818181818181817, "by_family": { "adversarial": { "correct": 6, "n": 12, "accuracy": 0.5 }, "ambiguous": { "correct": 6, "n": 14, "accuracy": 0.42857142857142855 }, "judge_hard": { "correct": 17, "n": 33, "accuracy": 0.5151515151515151 }, "long_policy": { "correct": 16, "n": 38, "accuracy": 0.42105263157894735 }, "multi_hop": { "correct": 9, "n": 35, "accuracy": 0.2571428571428571 }, "probability": { "correct": 9, "n": 20, "accuracy": 0.45 }, "routing_hard": { "correct": 0, "n": 10, "accuracy": 0.0 }, "temporal_numeric": { "correct": 9, "n": 30, "accuracy": 0.3 }, "tradeoff": { "correct": 3, "n": 12, "accuracy": 0.25 }, "trap": { "correct": 6, "n": 16, "accuracy": 0.375 } }, "has_distribution": true, "brier_mean": 0.6627383026655216, "ece": 0.022679951965887575, "probability_fidelity": 72.18897252368282, "calibration_score": 83.82649106525265, "onehot": { "ece": 0.6318181818181818, "probability_fidelity": 39.899499999999996, "calibration_score": 19.949749999999998 }, "latency_p50_s": 0.040032369550317526, "latency_p95_s": 0.24636488300748166, "mean_input_tokens": 4623.740909090909, "mean_output_tokens": 0.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 0.6763072957019215, "Balanced 33:33:33 (no calibration)": 0.5417823603513877, "Emphasis on Accuracy 60:20:20": 0.2737254447780234, "Emphasis on Speed 20:60:20": 0.7931605078498138, "Emphasis on Cost 20:20:60": 0.7324843168412952, "Intelligence only": 0.09829934724770434 }, "rank": 46, "rank_under": { "JevBench Score (25:25:25:25)": 46, "Balanced 33:33:33 (no calibration)": 46, "Emphasis on Accuracy 60:20:20": 46, "Emphasis on Speed 20:60:20": 46, "Emphasis on Cost 20:20:60": 46, "Intelligence only": 46 } }, { "key": "gte-reranker-modernbert-base", "display": "Alibaba GTE Reranker ModernBERT-base", "class": "reranker", "open": "yes", "author": "Alibaba-NLP", "repo": "https://huggingface.co/Alibaba-NLP/gte-reranker-modernbert-base", "licence": "Apache-2.0", "underlying": "149M ModernBERT cross-encoder", "has_distribution": true, "probability_source": [ "public_calibrated_reranker_softmax" ], "endpoint_condition": "our GPU (lium.io A6000 48 GB), serial, one option batch per decision", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 0.3333333333333333, "standard": 0.3958333333333333, "judge": 0.3013698630136986, "hard": 0.33636363636363636 }, "axes": { "intelligence": 4.569305273729187, "calibration": 76.84759720665565, "speed": 90.58799397367488, "cost": 69.64001789071294 }, "jevbench_score": 0.32219057759911496, "speed": { "p50_s_raw": 0.048107756301760674, "p95_s_raw": 0.10235980194993316, "p50_s_adjusted": 0.24621551260352134, "p95_s_adjusted": 0.35471960389986634, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.051893850322812796, "hard_tier_p95_s": 0.18488985453732257 }, "cost": { "kind": "measured", "usd_per_1000": 0.010280148864935354, "basis": "MEASURED model inference time x lium.io A6000 tariff USD 0.42/hour", "usd_per_1000_v11_tiers": 0.01057049607889699, "usd_per_1000_hard": 0.009865744205008289, "self_host_sensitivity": "Whole five-model rental including setup/download was USD 0.21." }, "calibration": { "score": 76.84759720665565, "score_label_only_as_onehot": null, "ece_hard": 0.0812949237783053, "probability_fidelity": 69.95417916897236, "brier_hard": 0.6771539067858408, "brier_standard_judge_v11": 0.805535942323157, "note": null }, "hard": { "run": "runs/gte-reranker-modernbert-base--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.33636363636363636, "by_family": { "adversarial": { "correct": 7, "n": 12, "accuracy": 0.5833333333333334 }, "ambiguous": { "correct": 7, "n": 14, "accuracy": 0.5 }, "judge_hard": { "correct": 15, "n": 33, "accuracy": 0.45454545454545453 }, "long_policy": { "correct": 11, "n": 38, "accuracy": 0.2894736842105263 }, "multi_hop": { "correct": 8, "n": 35, "accuracy": 0.22857142857142856 }, "probability": { "correct": 7, "n": 20, "accuracy": 0.35 }, "routing_hard": { "correct": 0, "n": 10, "accuracy": 0.0 }, "temporal_numeric": { "correct": 7, "n": 30, "accuracy": 0.23333333333333334 }, "tradeoff": { "correct": 4, "n": 12, "accuracy": 0.3333333333333333 }, "trap": { "correct": 8, "n": 16, "accuracy": 0.5 } }, "has_distribution": true, "brier_mean": 0.6771539067858408, "ece": 0.0812949237783053, "probability_fidelity": 69.95417916897236, "calibration_score": 76.84759720665565, "onehot": { "ece": 0.6636363636363636, "probability_fidelity": 34.50550000000001, "calibration_score": 17.252750000000006 }, "latency_p50_s": 0.051893850322812796, "latency_p95_s": 0.18488985453732257, "mean_input_tokens": 4257.440909090909, "mean_output_tokens": 0.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 0.32219057759911496, "Balanced 33:33:33 (no calibration)": 0.25606697421804375, "Emphasis on Accuracy 60:20:20": 0.11957939374841434, "Emphasis on Speed 20:60:20": 0.3949521975742231, "Emphasis on Cost 20:20:60": 0.35551655363419166, "Intelligence only": 0.03816018870025685 }, "rank": 47, "rank_under": { "JevBench Score (25:25:25:25)": 47, "Balanced 33:33:33 (no calibration)": 47, "Emphasis on Accuracy 60:20:20": 47, "Emphasis on Speed 20:60:20": 47, "Emphasis on Cost 20:20:60": 47, "Intelligence only": 47 } }, { "key": "certo", "display": "Certo v1 (AltSlate Labs)", "class": "jev-rebuild", "open": "yes", "author": "AltSlate Labs", "repo": "https://huggingface.co/altslate/certo-decision-model", "licence": "MIT", "underlying": "ModernBERT-large with a per-option query/scoring head, ~400M parameters", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "our RunPod GPU (GeForce RTX 3090 24 GB, community cloud), reached over the internet from Germany", "endpoint_kind": "gpu", "partial": false, "ranked": true, "listing": "ranked", "not_ranked_because": null, "tiers": { "easy": 0.2777777777777778, "standard": 0.3020833333333333, "judge": 0.2191780821917808, "hard": 0.3181818181818182 }, "axes": { "intelligence": 0.0, "calibration": 82.02971590909091, "speed": 93.97471093191521, "cost": 100.0 }, "jevbench_score": 0.0, "speed": { "p50_s_raw": 0.019205978140234947, "p95_s_raw": 0.031265050335787234, "p50_s_adjusted": 0.1884119562804699, "p95_s_adjusted": 0.21253010067157446, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.019906828412786126, "hard_tier_p95_s": 0.02713130824267863 }, "cost": { "kind": "estimate", "usd_per_1000": 0.0009654868913857679, "basis": "ESTIMATE: hosted-provider price, deepinfra encoders of the same size (bge-large, e5-large, Qwen3-Embedding-0.6B) list price $0.01/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 86 input and 0 output tokens per decision (input tokens measured (the system's own count))", "usd_per_1000_v11_tiers": 0.0008618471337579619, "usd_per_1000_hard": 0.001113409090909091, "self_host_sensitivity": null }, "calibration": { "score": 82.02971590909091, "score_label_only_as_onehot": null, "ece_hard": 0.037609090909090884, "probability_fidelity": 71.58125, "brier_hard": 0.6629653915000002, "brier_standard_judge_v11": 0.6887553576859503, "note": null }, "hard": { "run": "runs/certo--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.3181818181818182, "by_family": { "adversarial": { "correct": 2, "n": 12, "accuracy": 0.16666666666666666 }, "ambiguous": { "correct": 4, "n": 14, "accuracy": 0.2857142857142857 }, "judge_hard": { "correct": 15, "n": 33, "accuracy": 0.45454545454545453 }, "long_policy": { "correct": 13, "n": 38, "accuracy": 0.34210526315789475 }, "multi_hop": { "correct": 7, "n": 35, "accuracy": 0.2 }, "probability": { "correct": 7, "n": 20, "accuracy": 0.35 }, "routing_hard": { "correct": 2, "n": 10, "accuracy": 0.2 }, "temporal_numeric": { "correct": 12, "n": 30, "accuracy": 0.4 }, "tradeoff": { "correct": 2, "n": 12, "accuracy": 0.16666666666666666 }, "trap": { "correct": 6, "n": 16, "accuracy": 0.375 } }, "has_distribution": true, "brier_mean": 0.6629653915000002, "ece": 0.037609090909090884, "probability_fidelity": 71.58125, "calibration_score": 82.02971590909091, "onehot": { "ece": 0.6818181818181819, "probability_fidelity": 38.06900000000001, "calibration_score": 19.034500000000005 }, "latency_p50_s": 0.019906828412786126, "latency_p95_s": 0.02713130824267863, "mean_input_tokens": 111.3409090909091, "mean_output_tokens": 0.0, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 0.0, "Balanced 33:33:33 (no calibration)": 0.0, "Emphasis on Accuracy 60:20:20": 0.0, "Emphasis on Speed 20:60:20": 0.0, "Emphasis on Cost 20:20:60": 0.0, "Intelligence only": 0.0 }, "rank": 48, "rank_under": { "JevBench Score (25:25:25:25)": 48, "Balanced 33:33:33 (no calibration)": 48, "Emphasis on Accuracy 60:20:20": 48, "Emphasis on Speed 20:60:20": 48, "Emphasis on Cost 20:20:60": 48, "Intelligence only": 48 } }, { "key": "classifier-dev-fast", "display": "classifier.dev (fast tier)", "class": "jev-service", "open": "no", "author": "mrmps (@michael_chomsky)", "repo": "https://classifier.dev", "licence": "MIT (code); hosted service", "underlying": "Jev (TypeSafe) behind classifier.dev's zero-shot classification API", "has_distribution": true, "probability_source": [ "native" ], "endpoint_condition": "production API (classifier.dev, fast tier)", "endpoint_kind": "api", "partial": false, "ranked": false, "listing": "honorable_mention", "not_ranked_because": "runs on Jev (TypeSafe) — listed, not ranked", "tiers": { "easy": 1.0, "standard": 0.9895833333333334, "judge": 0.9726027397260274, "hard": 0.7045454545454546 }, "axes": { "intelligence": 85.13161053003458, "calibration": 77.8658124426079, "speed": 87.59107265834845, "cost": 84.31363764158988 }, "jevbench_score": 83.64670446057299, "speed": { "p50_s_raw": 0.38636084645986557, "p95_s_raw": 0.4507125232368707, "p50_s_adjusted": 0.38636084645986557, "p95_s_adjusted": 0.4507125232368707, "adjustment": "none (production API)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 0.38397014886140823, "hard_tier_p95_s": 0.45760550089180463 }, "cost": { "kind": "estimate", "usd_per_1000": 0.003333333333333333, "basis": "ESTIMATE from the published paid plan (the free tier was used): classifier.dev Pro $20/month for 200,000 fast classifications a day (https://classifier.dev/pricing, read 2026-09-19) = $0.0033 per 1,000 decisions at full use; one decision = one classification. Lower use costs more per decision: at a tenth of that allowance it is $0.033 per 1,000, and the free tier (20,000 fast classifications a day, which is what this run used) costs nothing.", "usd_per_1000_v11_tiers": 0.003333333333333333, "usd_per_1000_hard": 0.003333333333333333, "self_host_sensitivity": null }, "calibration": { "score": 77.8658124426079, "score_label_only_as_onehot": null, "ece_hard": 0.09111937557392094, "probability_fidelity": 73.9555, "brier_hard": 0.3604363282039868, "brier_standard_judge_v11": 0.04268640347881518, "note": null }, "hard": { "run": "runs/classifier-dev-fast--hard (job jevbench-add-requests-20260919)", "n_items": 220, "n_ok": 220, "n_attempted": 220, "coverage": 1.0, "success_rate": 1.0, "accuracy": 0.7045454545454546, "by_family": { "adversarial": { "correct": 12, "n": 12, "accuracy": 1.0 }, "ambiguous": { "correct": 11, "n": 14, "accuracy": 0.7857142857142857 }, "judge_hard": { "correct": 26, "n": 33, "accuracy": 0.7878787878787878 }, "long_policy": { "correct": 20, "n": 38, "accuracy": 0.5263157894736842 }, "multi_hop": { "correct": 28, "n": 35, "accuracy": 0.8 }, "probability": { "correct": 14, "n": 20, "accuracy": 0.7 }, "routing_hard": { "correct": 10, "n": 10, "accuracy": 1.0 }, "temporal_numeric": { "correct": 7, "n": 30, "accuracy": 0.23333333333333334 }, "tradeoff": { "correct": 11, "n": 12, "accuracy": 0.9166666666666666 }, "trap": { "correct": 16, "n": 16, "accuracy": 1.0 } }, "has_distribution": true, "brier_mean": 0.3604363282039868, "ece": 0.09111937557392094, "probability_fidelity": 73.9555, "calibration_score": 77.8658124426079, "onehot": { "ece": 0.2954545454545454, "probability_fidelity": 56.04099999999998, "calibration_score": 48.47504545454545 }, "latency_p50_s": 0.38397014886140823, "latency_p95_s": 0.45760550089180463, "mean_input_tokens": null, "mean_output_tokens": null, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 83.64670446057299, "Balanced 33:33:33 (no calibration)": 85.66751588458473, "Emphasis on Accuracy 60:20:20": 85.45275010322048, "Emphasis on Speed 20:60:20": 86.43181684291058, "Emphasis on Cost 20:20:60": 85.12337513924754, "Intelligence only": 85.13161053003458 }, "rank": null }, { "key": "qwen3.8-27b", "display": "Qwen3.8 27B (Chutes TEE)", "class": "llm-baseline", "open": "weights", "author": "Qwen / Chutes", "repo": null, "licence": "open weights", "underlying": "Qwen3.8-27B", "has_distribution": true, "probability_source": [ "verbalized" ], "endpoint_condition": "Chutes shared inference (TEE)", "endpoint_kind": "api", "partial": true, "ranked": false, "listing": "partial", "not_ranked_because": null, "tiers": { "easy": 0.9861111111111112, "standard": 0.9895833333333334, "judge": 0.952755905511811, "hard": 0.21363636363636362 }, "axes": { "intelligence": 67.4325524043019, "calibration": 92.09887121241667, "speed": 61.270489995780956, "cost": 0.0 }, "jevbench_score": 24.836695615027768, "speed": { "p50_s_raw": 5.753906108438969, "p95_s_raw": 12.97144114784895, "p50_s_adjusted": 5.753906108438969, "p95_s_adjusted": 12.97144114784895, "adjustment": "none (production API)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "from a Hetzner server in Germany, network included", "hard_tier_p50_s": 20.471472900360823, "hard_tier_p95_s": 74.47066541947424 }, "cost": { "kind": "estimate", "usd_per_1000": 2.669088141927965, "basis": "ESTIMATE: hosted-provider price, openrouter qwen/qwen3.8-27b list price $0.214/M in, $2.55/M out (same weights; our run used a flat-rate Chutes subscription) x 416 input and 393 output tokens per decision [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | ESTIMATE: openrouter qwen/qwen3.8-27b $0.214/M in, $2.55/M out x 1592 in / 1833 out tokens per hard decision", "usd_per_1000_v11_tiers": 1.0249957201365187, "usd_per_1000_hard": 5.0156564166666655, "self_host_sensitivity": { "usd_per_1000": 3.8023331395736286, "score": 10.498745882580529, "machine": "1x A100 80 GB (Qwen3.8-27B bf16)", "usd_per_h": 1.39, "concurrency": 4, "utilisation": 0.3, "p50_s_used": 11.81732313881876, "decisions_per_hour": 365.5650225734472 } }, "calibration": { "score": 92.09887121241667, "score_label_only_as_onehot": null, "ece_hard": 0.07900871212083338, "probability_fidelity": 99.999484849, "brier_hard": 0.07885637798841445, "brier_standard_judge_v11": 0.019061104047845712, "note": null }, "hard": { "run": "runs/qwen3.8-27b--hard-mt16k", "n_items": 220, "n_ok": 48, "n_attempted": 51, "coverage": 0.2318181818181818, "success_rate": 0.9411764705882353, "accuracy": 0.21363636363636362, "by_family": { "adversarial": { "correct": 0, "n": 12, "accuracy": 0.0 }, "ambiguous": { "correct": 7, "n": 14, "accuracy": 0.5 }, "judge_hard": { "correct": 0, "n": 33, "accuracy": 0.0 }, "long_policy": { "correct": 12, "n": 38, "accuracy": 0.3157894736842105 }, "multi_hop": { "correct": 5, "n": 35, "accuracy": 0.14285714285714285 }, "probability": { "correct": 10, "n": 20, "accuracy": 0.5 }, "routing_hard": { "correct": 0, "n": 10, "accuracy": 0.0 }, "temporal_numeric": { "correct": 7, "n": 30, "accuracy": 0.23333333333333334 }, "tradeoff": { "correct": 6, "n": 12, "accuracy": 0.5 }, "trap": { "correct": 0, "n": 16, "accuracy": 0.0 } }, "has_distribution": true, "brier_mean": 0.07885637798841445, "ece": 0.07900871212083338, "probability_fidelity": 99.999484849, "calibration_score": 92.09887121241667, "onehot": { "ece": 0.02083333333333337, "probability_fidelity": 66.726, "calibration_score": 81.27966666666666 }, "latency_p50_s": 20.471472900360823, "latency_p95_s": 74.47066541947424, "mean_input_tokens": 1591.6041666666667, "mean_output_tokens": 1833.3541666666667, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 24.836695615027768, "Balanced 33:33:33 (no calibration)": 16.046253564707456, "Emphasis on Accuracy 60:20:20": 28.495221017399626, "Emphasis on Speed 20:60:20": 27.423616805375097, "Emphasis on Cost 20:20:60": 5.287181148877906, "Intelligence only": 67.43255240430189 }, "rank": null }, { "key": "needle-3-tools", "display": "Needle 3, options as tools (post-hoc adapter mode)", "class": "small-tool-model", "open": "yes", "author": "Cactus Compute", "repo": "https://github.com/cactus-compute/needle", "licence": "Apache-2.0 (model and package)", "underlying": "Needle 3 (121M parameters, 2-bit)", "has_distribution": false, "probability_source": [ "label_only_no_calibrated_distribution" ], "endpoint_condition": "our CPU (2 threads, Ryzen 5 3600)", "endpoint_kind": "cpu", "partial": true, "ranked": false, "listing": "partial", "not_ranked_because": null, "tiers": { "easy": 0.6666666666666666, "standard": 0.3125, "judge": 0.3424657534246575, "hard": null }, "axes": { "intelligence": 13.527361471793748, "calibration": null, "speed": 52.842525649949984, "cost": 65.27447796986436 }, "jevbench_score": 1.0757743846539873, "speed": { "p50_s_raw": 3.778900783509016, "p95_s_raw": 33.63718595951795, "p50_s_adjusted": 7.707801567018032, "p95_s_adjusted": 67.4243719190359, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "local, 2 CPU threads of a Ryzen 5 3600", "hard_tier_p50_s": null, "hard_tier_p95_s": null }, "cost": { "kind": "estimate", "usd_per_1000": 0.014372006369426754, "basis": "ESTIMATE: same per-token price as Needle 3 (openrouter meta-llama/llama-3.2-1b-instruct $0.027/M in, $0.201/M out) x 383 input and 20 output tokens per decision, over the 314 easy/standard/judge decisions it ran (no hard-tier run). The v1.2 score lab had no price for this row and scored it 100; fixed. [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json]", "usd_per_1000_v11_tiers": 0.014372006369426754, "usd_per_1000_hard": null, "self_host_sensitivity": null }, "calibration": { "score": null, "score_label_only_as_onehot": null, "ece_hard": null, "probability_fidelity": null, "brier_hard": null, "brier_standard_judge_v11": null, "note": "returns a label, not a probability distribution: no calibration score (counts as 0 in the JevBench Score)" }, "hard": null, "presets": { "JevBench Score (25:25:25:25)": 1.0757743846539873, "Balanced 33:33:33 (no calibration)": 2.635114787619102, "Emphasis on Accuracy 60:20:20": 1.781391737635823, "Emphasis on Speed 20:60:20": 3.072328307511425, "Emphasis on Cost 20:20:60": 3.343270825065677, "Intelligence only": 0.9901460902072077 }, "rank": null }, { "key": "needle-3", "display": "Needle 3 (Cactus, 2-bit, local CPU)", "class": "small-tool-model", "open": "yes", "author": "Cactus Compute", "repo": "https://github.com/cactus-compute/needle", "licence": "Apache-2.0 (model and package)", "underlying": "Needle 3 (121M parameters, 2-bit)", "has_distribution": false, "probability_source": [ "label_only_no_calibrated_distribution" ], "endpoint_condition": "our CPU (2 threads, Ryzen 5 3600)", "endpoint_kind": "cpu", "partial": true, "ranked": false, "listing": "partial", "not_ranked_because": null, "tiers": { "easy": 0.4722222222222222, "standard": 0.16666666666666666, "judge": 0.3150684931506849, "hard": 0.07727272727272727 }, "axes": { "intelligence": 4.583148211883233, "calibration": null, "speed": 59.92368509307548, "cost": 58.68121468207322 }, "jevbench_score": 0.09466799356370086, "speed": { "p50_s_raw": 1.6870602630078793, "p95_s_raw": 14.364453018829225, "p50_s_adjusted": 3.5241205260157584, "p95_s_adjusted": 28.87890603765845, "adjustment": "x2 + 0.15 s (assumption, not measured)", "run": "serial 242-decision standard+judge run", "hardware": null, "measured_where": "local, 2 CPU threads of a Ryzen 5 3600", "hard_tier_p50_s": 19.284176252782345, "hard_tier_p95_s": 155.37855613604194 }, "cost": { "kind": "estimate", "usd_per_1000": 0.023839264044943822, "basis": "ESTIMATE: hosted-provider price, openrouter meta-llama/llama-3.2-1b-instruct list price $0.027/M in, $0.201/M out (no generative model under 1B is listed; the smallest listed one (1B) errs high; about 20 generated tokens for one tool call) x 383 input and 20 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | ESTIMATE: openrouter meta-llama/llama-3.2-1b-instruct $0.027/M in, $0.201/M out x 1235 in / 20 out tokens per hard decision", "usd_per_1000_v11_tiers": 0.014372006369426754, "usd_per_1000_hard": 0.03735162272727273, "self_host_sensitivity": { "usd_per_1000": 0.059578722823927475, "score": 55.62272027242932, "machine": "Hetzner CX22 (2 vCPU)", "usd_per_h": 0.0072, "concurrency": 1, "utilisation": 0.3, "p50_s_used": 8.93680842358912, "decisions_per_hour": 120.84851199778322 } }, "calibration": { "score": null, "score_label_only_as_onehot": 22.869444444444444, "ece_hard": null, "probability_fidelity": null, "brier_hard": null, "brier_standard_judge_v11": null, "note": "returns a label, not a probability distribution: no calibration score (counts as 0 in the JevBench Score)" }, "hard": { "run": "runs/needle-3--hard", "n_items": 220, "n_ok": 44, "n_attempted": 44, "coverage": 0.2, "success_rate": 1.0, "accuracy": 0.07727272727272727, "by_family": { "adversarial": { "correct": 0, "n": 12, "accuracy": 0.0 }, "ambiguous": { "correct": 3, "n": 14, "accuracy": 0.21428571428571427 }, "judge_hard": { "correct": 0, "n": 33, "accuracy": 0.0 }, "long_policy": { "correct": 5, "n": 38, "accuracy": 0.13157894736842105 }, "multi_hop": { "correct": 0, "n": 35, "accuracy": 0.0 }, "probability": { "correct": 4, "n": 20, "accuracy": 0.2 }, "routing_hard": { "correct": 0, "n": 10, "accuracy": 0.0 }, "temporal_numeric": { "correct": 3, "n": 30, "accuracy": 0.1 }, "tradeoff": { "correct": 2, "n": 12, "accuracy": 0.16666666666666666 }, "trap": { "correct": 0, "n": 16, "accuracy": 0.0 } }, "has_distribution": false, "brier_mean": null, "ece": null, "probability_fidelity": null, "calibration_score": null, "onehot": { "ece": 0.6136363636363636, "probability_fidelity": 45.73888888888889, "calibration_score": 22.869444444444444 }, "latency_p50_s": 19.284176252782345, "latency_p95_s": 155.37855613604194, "mean_input_tokens": null, "mean_output_tokens": null, "charged_usd": 0.0 }, "presets": { "JevBench Score (25:25:25:25)": 0.09466799356370086, "Balanced 33:33:33 (no calibration)": 0.21223074491406504, "Emphasis on Accuracy 60:20:20": 0.1072273709369308, "Emphasis on Speed 20:60:20": 0.2998330498245854, "Emphasis on Cost 20:20:60": 0.2973306875264691, "Intelligence only": 0.03850806506674241 }, "rank": null } ] }