{
  "schema": 1,
  "snapshot_date": "2026-09-23",
  "official": {
    "reported_revision": "v1.3.0",
    "commit": "75e6224ed8103bbc3485ca74820a2eaf7ce8abe0",
    "source_sha256": "20fce8e6e4f078d4a3d0105abb5739bacabbb6d4112a1f8be7d2db76f44cdb78",
    "source_url": "https://github.com/fstandhartinger/jevbench/blob/75e6224ed8103bbc3485ca74820a2eaf7ce8abe0/results/v1.2/jevbench-v1.2-results.json",
    "score_method": "Geometric mean of Intelligence, Calibration, Speed and Cost, 25 % each. If chance-corrected Intelligence is below 50, multiply by (Intelligence / 50)^2; at or above 50 there is no penalty.",
    "score_provenance": "Published numeric fields copied verbatim; not recomputed.",
    "accuracy_method": "Raw correct counts reconstructed from published tier accuracies × tier sizes; unweighted over 534 tasks.",
    "models": [
      {
        "key": "jev-1.13.0",
        "name": "Jev 1.13.0 (TypeSafe AI)",
        "rank": 1,
        "score": 74.40448849535433,
        "correct": 468,
        "n": 534,
        "accuracy_pct": 87.64044943820225,
        "cost_usd_per_1000": 0.03991423595505617,
        "cost_kind": "measured",
        "cost_basis": "public tariff x measured tokens (https://docs.typesafe.ai/models (output tokens not billed)) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | public tariff x measured tokens (hard-tier run)",
        "p50_s": 0.6524335257709026,
        "p95_s": 0.7221905551850795,
        "adjusted_p50_s": 0.6524335257709026,
        "adjusted_p95_s": 0.7221905551850795,
        "latency_adjustment": "none (production API)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "production API (api.typesafe.ai)",
        "repo": "https://docs.typesafe.ai",
        "note": ""
      },
      {
        "key": "semif-qwen3.5-4b",
        "name": "SemIf, formerly OpenJev (Qwen3.5-4B, TheoLeeCJ)",
        "rank": 2,
        "score": 73.08747495585776,
        "correct": 436,
        "n": 534,
        "accuracy_pct": 81.64794007490637,
        "cost_usd_per_1000": 0.02244460674157303,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, deepinfra Qwen/Qwen3.5-4B list price $0.03/M in, $0.15/M out (same weights (not on OpenRouter), as open-alternative-jev in v1.1.2) x 396 input and 1 output tokens per decision (input tokens measured) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | ESTIMATE: deepinfra Qwen/Qwen3.5-4B $0.03/M in, $0.15/M out x 1244 in / 0 out tokens per hard decision",
        "p50_s": 0.19796114787459373,
        "p95_s": 0.3153164997696876,
        "adjusted_p50_s": 0.5459222957491875,
        "adjusted_p95_s": 0.7806329995393753,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": "RunPod RTX PRO 4500 Blackwell 32 GB (EU-RO-1)",
        "measured_where": "from a Hetzner server in Germany over the internet to the pod's public TCP port (plain HTTP, one connection per request) through a thin transport around the author's library; model loaded before timing",
        "endpoint": "our RunPod GPU (RTX PRO 4500 Blackwell 32 GB (EU-RO-1)), reached over the internet",
        "repo": "https://github.com/TheoLeeCJ/openjev",
        "note": ""
      },
      {
        "key": "djev",
        "name": "djev (Maisa, diffusion-gemma)",
        "rank": 3,
        "score": 73.02927314568292,
        "correct": 455,
        "n": 534,
        "accuracy_pct": 85.2059925093633,
        "cost_usd_per_1000": 0.025951254681647943,
        "cost_kind": "announced",
        "cost_basis": "ANNOUNCED PRICE (free preview): djev's docs state $0.035 per million input tokens, output tokens free (https://api.djev.dev/docs, 'Usage & credits'; prepaid billing not yet switched on, 19 Sep 2026, so nothing was charged) x measured input tokens (741 per decision on average over all 534 decisions)",
        "p50_s": 0.2370578795671463,
        "p95_s": 0.30865143015980717,
        "adjusted_p50_s": 0.2370578795671463,
        "adjusted_p95_s": 0.30865143015980717,
        "latency_adjustment": "none (production API)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "production API (api.djev.dev, free preview)",
        "repo": "https://github.com/Davipar/djev-dev",
        "note": "The measured endpoint was Maisa's hosted API in free preview; the cost uses its announced price ($0.035 per million input tokens, output free), and nothing was charged. The self-hostable djev-dev runtime is Apache-2.0 and applies a structured one-step inference method to Google's Apache-2.0 diffusiongemma-26B-A4B-it checkpoint; it adds no separately trained djev weights. Probabilities are djev's own (its docs call them experimental and uncalibrated)."
      },
      {
        "key": "winnow-12b",
        "name": "Winnow-12B Q8",
        "rank": 4,
        "score": 71.21883011371422,
        "correct": 454,
        "n": 534,
        "accuracy_pct": 85.0187265917603,
        "cost_usd_per_1000": 0.037087359550561805,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, OpenRouter google/gemma-3-12b-it hosted reference list price $0.05/M in, $0.0/M out (the nearest publicly hosted 12B Gemma sibling; Winnow reads answer logits in one forward pass and generates no answer tokens) x 393 input and 0 output tokens per decision (input tokens measured (the system's own count))",
        "p50_s": 0.22503555566072464,
        "p95_s": 0.4126893173903227,
        "adjusted_p50_s": 0.6000711113214493,
        "adjusted_p95_s": 0.9753786347806455,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our GPU (lium.io RTX 4090 24 GB), reached over the internet from Germany; serial, one request at a time",
        "repo": "https://huggingface.co/EldanRing/Winnow-12B",
        "note": "The submitted Q8_0 GGUF ran through the pinned author's TypeSafe-compatible /v1/systemone server with 8,192 context, four resident decision branches, Q8 KV, and full GPU offload. The private training corpus was not released. The author's checksum-based audit reports zero exact public-item overlap, but that claim cannot be independently reproduced; our scan found no exact public state or instruction text in the released artifacts. Cost uses the $0.05/M-input hosted Gemma 3 12B reference, not free/100."
      },
      {
        "key": "reflex-4b",
        "name": "reflex 4B (kshetrajna12)",
        "rank": 5,
        "score": 70.31658532999862,
        "correct": 444,
        "n": 534,
        "accuracy_pct": 83.14606741573034,
        "cost_usd_per_1000": 0.022087303370786515,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, DeepInfra Qwen/Qwen3.5-4B list price $0.03/M in, $0.0/M out (the exact base weights; one pass, no generated output) x 377 input and 0 output tokens per decision (input tokens measured (the system's own count))",
        "p50_s": 1.7996779419481754,
        "p95_s": 2.0541166696697473,
        "adjusted_p50_s": 3.7493558838963508,
        "adjusted_p95_s": 4.258233339339495,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our RunPod GPU (H100 NVL 96 GB, Canada), reached over the internet from Germany",
        "repo": "https://github.com/kshetrajna12/reflex",
        "note": "The author's reflex-serve: Qwen3.5-4B with the published LoRA and its per-primitive calibration file; the state is encoded once and each question read from the label logits. Run serially on our GPU; the author discloses that the 231 public items were used four times as a development gate."
      },
      {
        "key": "jqv",
        "name": "jqv (Qwen3-32B zero-shot)",
        "rank": 6,
        "score": 68.62862396499413,
        "correct": 441,
        "n": 534,
        "accuracy_pct": 82.58426966292134,
        "cost_usd_per_1000": 0.05641543071161049,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, OpenRouter qwen/qwen3-32b list price $0.08/M in, $0.0/M out (the exact base model this system reads logits from; nothing is generated) x 359 input and 0 output tokens per decision (input tokens measured (the system's own count))",
        "p50_s": 0.7473905384540558,
        "p95_s": 0.9735059145838022,
        "adjusted_p50_s": 1.6447810769081115,
        "adjusted_p95_s": 2.0970118291676045,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our RunPod GPU (H100 NVL 96 GB, Canada), reached over the internet from Germany",
        "repo": "https://github.com/Octalab-Inc/jqv",
        "note": "A stock Qwen3-32B with no decision training: the state is prefilled once, each question is an isolated branch and the answer is read from the option-letter logits, with one fitted temperature (3.02, 400 MMLU validation items). Re-run in v1.2.8 on our own GPU from the now-public serving code (Octalab-Inc/jqv 0189b67), so all 534 decisions including the held-out hard items were asked; this full run replaces the v1.2.7 partial row, which had been measured on the submitter's machine. Cost is the base model's public per-token tariff, not free."
      },
      {
        "key": "decision-machine-1",
        "name": "decision-machine-1 (milliseconds.ai)",
        "rank": 7,
        "score": 68.33662413233554,
        "correct": 379,
        "n": 534,
        "accuracy_pct": 70.97378277153558,
        "cost_usd_per_1000": 0.035026591760299625,
        "cost_kind": "measured",
        "cost_basis": "public tariff x measured tokens: $0.04 per million input tokens, output free (https://docs.milliseconds.ai/reference/pricing, read 2026-09-21) x 496 input tokens per easy/standard/judge decision as reported by the API; the run used the free test key, the price is the paid one",
        "p50_s": 0.17223640158772469,
        "p95_s": 0.29605407454073424,
        "adjusted_p50_s": 0.17223640158772469,
        "adjusted_p95_s": 0.29605407454073424,
        "latency_adjustment": "none (production API)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "production API (milliseconds.ai, served from its nearest region), measured from Germany",
        "repo": "https://www.milliseconds.ai",
        "note": "A closed-weights decision model behind a production API that serves TypeSafe's wire format, so the unchanged typesafe adapter ran it. Run on a free test key (30 requests a minute, 2.2 s between requests); the provider states the inference infrastructure is the same as for paid keys. Cost is the public paid tariff, $0.04 per million input tokens (output free), times the input tokens the API reported."
      },
      {
        "key": "decider-35b-a3b",
        "name": "decider-35b-a3b (Mapika)",
        "rank": 8,
        "score": 67.55405352545425,
        "correct": 442,
        "n": 534,
        "accuracy_pct": 82.77153558052434,
        "cost_usd_per_1000": 0.0665503745318352,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, OpenRouter Qwen3.6-35B-A3B list price list price $0.1/M in, $0.0/M out (the closest public hosted 35B-A3B direct-logit model; no output is generated) x 312 input and 0 output tokens per decision (input tokens measured (the system's own count))",
        "p50_s": 0.29191894084215164,
        "p95_s": 0.49371074326336223,
        "adjusted_p50_s": 0.7338378816843033,
        "adjusted_p95_s": 1.1374214865267245,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our RunPod GPU (H100 NVL 96 GB), reached over the internet",
        "repo": "https://huggingface.co/Mapika/decider-35b-a3b",
        "note": "The author's TypeSafe-compatible server and published FP8 weights, run serially on our H100 NVL. The exhaustive startup batch warmup was skipped; each required serial shape captured lazily before its measured request. Self-host latency receives the standard ×2 + 0.15 s adjustment. Cost uses the closest hosted 35B-A3B input tariff and is not the temporary rental charge."
      },
      {
        "key": "open-alternative-jev",
        "name": "open-alternative-jev (Qwen3.5-4B, IkerMoel)",
        "rank": 9,
        "score": 66.9883439146374,
        "correct": 387,
        "n": 534,
        "accuracy_pct": 72.47191011235955,
        "cost_usd_per_1000": 0.022171404494382024,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, deepinfra Qwen/Qwen3.5-4B list price $0.03/M in, $0.15/M out (as open-alternative-jev) x 383 input and 1 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts) | ESTIMATE: deepinfra Qwen/Qwen3.5-4B $0.03/M in, $0.15/M out x 1235 in / 1 out tokens per hard decision",
        "p50_s": 0.20686038956046104,
        "p95_s": 0.32322231084108355,
        "adjusted_p50_s": 0.5637207791209221,
        "adjusted_p95_s": 0.7964446216821671,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": "RunPod RTX PRO 4500 Blackwell 32 GB (EU-RO-1)",
        "measured_where": "as open-alternative-jev",
        "endpoint": "our RunPod GPU (RTX PRO 4500 Blackwell 32 GB (EU-RO-1)), reached over the internet",
        "repo": "https://github.com/ikermoel/open-alternative-jev",
        "note": "With the options in reverse order (A. no, B. yes) the same model scored 21 % instead of 72 % on yes/no answer-judging items — small models are very sensitive to option order."
      },
      {
        "key": "system-one-open",
        "name": "system-one-open (Gemma 4 E2B LoRA on an L4)",
        "rank": 10,
        "score": 66.59745356122706,
        "correct": 398,
        "n": 534,
        "accuracy_pct": 74.53183520599251,
        "cost_usd_per_1000": 0.014880936329588014,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, deepinfra google/gemma-4-E4B-it list price $0.02/M in, $0.1/M out (Gemma 4 E2B is not listed; the nearest larger sibling, Gemma 4 E4B, is listed only on DeepInfra) x 383 input and 2 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | ESTIMATE: deepinfra google/gemma-4-E4B-it $0.02/M in, $0.1/M out x 1235 in / 2 out tokens per hard decision",
        "p50_s": 0.6517308317124844,
        "p95_s": 0.7724301926791667,
        "adjusted_p50_s": 1.3034616634249687,
        "adjusted_p95_s": 1.5448603853583334,
        "latency_adjustment": "x2 (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "author's public demo endpoint (Modal, L4) — not a production service",
        "repo": "https://github.com/mithalouni/system-one-open",
        "note": ""
      },
      {
        "key": "openjev-razorback16",
        "name": "OpenJev (DiffusionGemma 26B-A4B NVFP4, razorback16)",
        "rank": 11,
        "score": 66.36329072785742,
        "correct": 441,
        "n": 534,
        "accuracy_pct": 82.58426966292134,
        "cost_usd_per_1000": 0.06560455056179774,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, openrouter google/gemma-4-26b-a4b-it list price $0.09/M in, $0.3/M out (DiffusionGemma 26B-A4B is not listed; the same-size Gemma 4 26B-A4B MoE sibling is (size class moe_26B-A4B)) x 380 input and 1 output tokens per decision (input tokens measured) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | ESTIMATE: openrouter google/gemma-4-26b-a4b-it $0.09/M in, $0.3/M out x 1222 in / 0 out tokens per hard decision",
        "p50_s": 0.24127069488167763,
        "p95_s": 0.3052692499011755,
        "adjusted_p50_s": 0.6325413897633553,
        "adjusted_p95_s": 0.760538499802351,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": "RunPod RTX PRO 4500 Blackwell 32 GB (EU-RO-1)",
        "measured_where": "from a Hetzner server in Germany over the internet to the pod's public TCP port (plain HTTP, one connection per request); model loaded before timing",
        "endpoint": "our RunPod GPU (RTX PRO 4500 Blackwell 32 GB (EU-RO-1)), reached over the internet",
        "repo": "https://github.com/razorback16/openjev",
        "note": ""
      },
      {
        "key": "simplejev-qwen3.8-27b",
        "name": "SimpleJev Qwen3.8-27B",
        "rank": 12,
        "score": 66.29860868418298,
        "correct": 466,
        "n": 534,
        "accuracy_pct": 87.26591760299625,
        "cost_usd_per_1000": 0.10399651685393257,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, OpenRouter Gemma 4 26B-A4B size-class reference list price $0.09/M in, $0.0/M out (a public 27B dense model served as a direct-logit classifier; no output is generated) x 809 input and 0 output tokens per decision (input tokens measured (the system's own count))",
        "p50_s": 1.0138170085847378,
        "p95_s": 1.8794299442321063,
        "adjusted_p50_s": 2.0276340171694756,
        "adjusted_p95_s": 3.7588598884642126,
        "latency_adjustment": "x2 (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "author's public demo endpoint (Featherless Classifier Demo) — not a production service",
        "repo": "https://github.com/featherless-ai/simple-jev",
        "note": "Author's no-login shared demo, model id recorded verbatim, one request at a time at or below its 2 RPS limit. SimpleJev reads answer-token logits and returns the complete distribution; it does not generate an answer. Speed uses the public-demo x2 load adjustment; cost uses a hosted size-class input price and is not free/100."
      },
      {
        "key": "zerank-2",
        "name": "ZeroEntropy zerank-2",
        "rank": 13,
        "score": 65.96713617823673,
        "correct": 381,
        "n": 534,
        "accuracy_pct": 71.34831460674157,
        "cost_usd_per_1000": 0.04729792435014604,
        "cost_kind": "measured",
        "cost_basis": "MEASURED model inference time x lium.io A6000 tariff USD 0.42/hour",
        "p50_s": 0.12663885252550244,
        "p95_s": 1.5002395501825958,
        "adjusted_p50_s": 0.4032777050510049,
        "adjusted_p95_s": 3.1504791003651915,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our GPU (lium.io A6000 48 GB), serial, one option batch per decision",
        "repo": "https://huggingface.co/zeroentropy/zerank-2-reranker",
        "note": "Neutral documented reranker adapter; instruction and no-instruction public calibration were run, then frozen before one held-out pass."
      },
      {
        "key": "gpt-5.6-luna",
        "name": "GPT-5.6 Luna (low reasoning effort)",
        "rank": 14,
        "score": 65.94418818052425,
        "correct": 515,
        "n": 534,
        "accuracy_pct": 96.44194756554307,
        "cost_usd_per_1000": 0.24191123595505612,
        "cost_kind": "measured",
        "cost_basis": "public tariff x measured tokens (https://platform.openai.com/docs/pricing (standard tier, read 2026-09-19)) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | public tariff x measured tokens (hard-tier run)",
        "p50_s": 0.9680032916367054,
        "p95_s": 1.8175220962613816,
        "adjusted_p50_s": 0.9680032916367054,
        "adjusted_p95_s": 1.8175220962613816,
        "latency_adjustment": "none (production API)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "production API (OpenAI), reasoning effort low",
        "repo": null,
        "note": ""
      },
      {
        "key": "openjev-sglang",
        "name": "openjev-sglang (Qwen3.6-35B-A3B on SGLang)",
        "rank": 15,
        "score": 65.26826354532687,
        "correct": 460,
        "n": 534,
        "accuracy_pct": 86.14232209737828,
        "cost_usd_per_1000": 0.13127546816479402,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, openrouter qwen/qwen3.6-35b-a3b list price $0.1/M in, $0.9/M out (same base weights) x 610 input and 2 output tokens per decision [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | ESTIMATE: openrouter qwen/qwen3.6-35b-a3b $0.1/M in, $0.9/M out x 2272 in / 2 out tokens per hard decision",
        "p50_s": 0.6776718497276306,
        "p95_s": 0.7260066717863083,
        "adjusted_p50_s": 1.3553436994552612,
        "adjusted_p95_s": 1.4520133435726166,
        "latency_adjustment": "x2 (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "author's public demo endpoint (Modal) — not a production service",
        "repo": "https://github.com/ekzhang/openjev-sglang",
        "note": ""
      },
      {
        "key": "qwen3-reranker-4b",
        "name": "Qwen3-Reranker-4B",
        "rank": 16,
        "score": 63.82721228923758,
        "correct": 386,
        "n": 534,
        "accuracy_pct": 72.28464419475655,
        "cost_usd_per_1000": 0.04953623422454652,
        "cost_kind": "measured",
        "cost_basis": "MEASURED model inference time x lium.io A6000 tariff USD 0.42/hour",
        "p50_s": 0.1302479188889265,
        "p95_s": 1.5593349148519338,
        "adjusted_p50_s": 0.41049583777785303,
        "adjusted_p95_s": 3.2686698297038674,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our GPU (lium.io A6000 48 GB), serial, one option batch per decision",
        "repo": "https://huggingface.co/Qwen/Qwen3-Reranker-4B",
        "note": "Neutral documented reranker adapter; instruction and no-instruction public calibration were run, then frozen before one held-out pass."
      },
      {
        "key": "reflex-27b",
        "name": "reflex-27b (Qwen3.8-27B)",
        "rank": 17,
        "score": 63.33315017675085,
        "correct": 471,
        "n": 534,
        "accuracy_pct": 88.20224719101124,
        "cost_usd_per_1000": 0.18111733707865169,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, OpenRouter Qwen3.8-27B list price list price $0.214/M in, $0.0/M out (the exact public base weights used as a direct-logit classifier; no output is generated) x 481 input and 0 output tokens per decision (input tokens measured (the system's own count))",
        "p50_s": 1.8879061974585056,
        "p95_s": 2.2076592873781915,
        "adjusted_p50_s": 3.925812394917011,
        "adjusted_p95_s": 4.565318574756383,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our RunPod GPU (H100 NVL 96 GB), reached over the internet",
        "repo": "https://github.com/kshetrajna12/reflex",
        "note": "The frozen public Qwen3.8-27B checkpoint through reflex at the requested pinned commit, with two option orders averaged and temperature 1. No adapter or fitted calibration file. Run serially on our H100 NVL. Self-host latency receives the standard ×2 + 0.15 s adjustment; cost uses the exact base model's public hosted input tariff."
      },
      {
        "key": "litjev",
        "name": "LitJev (Qwen3.8-27B)",
        "rank": 18,
        "score": 62.69075853252832,
        "correct": 456,
        "n": 534,
        "accuracy_pct": 85.39325842696628,
        "cost_usd_per_1000": 0.16303594007490638,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, OpenRouter Qwen3.8-27B (as the reflex-27b row) list price $0.214/M in, $0.0/M out (the exact base weights; nothing is generated) x 418 input and 0 output tokens per decision (input tokens measured (the system's own count))",
        "p50_s": 2.025244139134884,
        "p95_s": 2.4555754292756315,
        "adjusted_p50_s": 4.200488278269768,
        "adjusted_p95_s": 5.0611508585512635,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our RunPod GPU (H100 NVL 96 GB, Canada), reached over the internet from Germany",
        "repo": "https://github.com/zhengxuyu/litjev",
        "note": "The author's reproduction of Jev's decision layer on an off-the-shelf model, in its default configuration: Qwen3.8-27B, scores read from the output head, no training and no calibration file (its README says probabilities are not calibrated by default). Run serially on our GPU through an SSH tunnel, because its server binds to localhost; the request still crosses the internet and gets the ×2 + 0.15 s adjustment."
      },
      {
        "key": "kev-0.6b",
        "name": "kev 0.6B (research preview)",
        "rank": 19,
        "score": 62.489006255662105,
        "correct": 335,
        "n": 534,
        "accuracy_pct": 62.734082397003746,
        "cost_usd_per_1000": 0.0062676217228464426,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, DeepInfra Qwen3-Embedding-0.6B size-class reference list price $0.01/M in, $0.0/M out (a <=0.6B one-pass model with no generated output) x 279 input and 0 output tokens per decision (input tokens measured (the system's own count))",
        "p50_s": 0.5904003903269768,
        "p95_s": 0.9702187780290842,
        "adjusted_p50_s": 1.3308007806539535,
        "adjusted_p95_s": 2.0904375560581685,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our RunPod GPU (GeForce RTX 3090 24 GB, community cloud CA), reached over the internet",
        "repo": "https://github.com/jaredpalmer/kev",
        "note": "Self-hosted from the author's repository at commit 20fa626 through its native TypeSafe-compatible `/v1/systemone` server, BF16 on an RTX 3090; measured serially from Sandy over the internet. The author labels this checkpoint a research preview."
      },
      {
        "key": "simplejev-qwen3.6-35b-a3b",
        "name": "SimpleJev Qwen3.6-35B-A3B",
        "rank": 20,
        "score": 62.48305419264297,
        "correct": 444,
        "n": 534,
        "accuracy_pct": 83.14606741573034,
        "cost_usd_per_1000": 0.11555168539325843,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, OpenRouter Qwen3.6-35B-A3B list price list price $0.1/M in, $0.0/M out (the same base weights served as a direct-logit classifier; no output is generated) x 809 input and 0 output tokens per decision (input tokens measured (the system's own count))",
        "p50_s": 0.8519573211669922,
        "p95_s": 0.9304875515401363,
        "adjusted_p50_s": 1.7039146423339844,
        "adjusted_p95_s": 1.8609751030802726,
        "latency_adjustment": "x2 (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "author's public demo endpoint (Featherless Classifier Demo) — not a production service",
        "repo": "https://github.com/featherless-ai/simple-jev",
        "note": "Author's no-login shared demo, model id recorded verbatim, one request at a time at or below its 2 RPS limit. SimpleJev reads answer-token logits and returns the complete distribution; it does not generate an answer. Speed uses the public-demo x2 load adjustment; cost uses a hosted size-class input price and is not free/100."
      },
      {
        "key": "djev-thinking",
        "name": "djev (thinking)",
        "rank": 21,
        "score": 62.35960888483864,
        "correct": 452,
        "n": 534,
        "accuracy_pct": 84.6441947565543,
        "cost_usd_per_1000": 0.27429398876404487,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: same-size hosted reference x 749 measured input and 690 measured output tokens per attempted decision across all 534, failures included",
        "p50_s": 0.4260098780505359,
        "p95_s": 1.448969176481475,
        "adjusted_p50_s": 1.0020197561010717,
        "adjusted_p95_s": 3.04793835296295,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our GPU (lium.io H200 141 GB), reached over the internet from Germany; serial, one request at a time",
        "repo": "https://github.com/Davipar/djev-dev",
        "note": "Experimental full-generation path over the same DiffusionGemma checkpoint as djev-dev: thinking was enabled and the model could generate up to 8,192 tokens before returning its distribution. Current djev-dev itself hard-codes enable_thinking=false, diffusion_max_steps=1 and read_only=true, so this is not a switch in its published typed API. It is substantially slower/costlier, and 72/534 requests exhausted the output budget without a parseable distribution; those are failures. Cost uses measured tokens and a same-size hosted reference, not the H200 rental bill."
      },
      {
        "key": "jev-local",
        "name": "jev-local (Qwen3.5-9B)",
        "rank": 22,
        "score": 61.79680786600313,
        "correct": 413,
        "n": 534,
        "accuracy_pct": 77.34082397003745,
        "cost_usd_per_1000": 0.07745842696629213,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, OpenRouter qwen/qwen3.5-9b list price $0.1/M in, $0.0/M out (the exact base weights; scored by log-probabilities, nothing is generated) x 452 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts)",
        "p50_s": 1.0450106970965862,
        "p95_s": 2.6157593585550782,
        "adjusted_p50_s": 2.2400213941931724,
        "adjusted_p95_s": 5.381518717110157,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our RunPod GPU (H100 NVL 96 GB, Canada), reached over the internet from Germany",
        "repo": "https://github.com/us/jev-local",
        "note": "The author's local Jev-compatible server in its default full configuration: a frozen Qwen3.5-9B scores each option by its mean log-probability (one forward pass per option, no generation, no decision training). Run serially on our GPU. It re-reads the state once per option; if its reported token count covers one pass only, a per-token hosted price would be higher than this estimate."
      },
      {
        "key": "decider-2b",
        "name": "decider-2b (Mapika)",
        "rank": 23,
        "score": 61.6816875223098,
        "correct": 371,
        "n": 534,
        "accuracy_pct": 69.47565543071161,
        "cost_usd_per_1000": 0.01996511235955056,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, DeepInfra Qwen/Qwen3.5-4B list price $0.03/M in, $0.0/M out (no hosted ~2B Qwen3.5 is listed, so the 4B price is used and errs high; one pass, no output) x 312 input and 0 output tokens per decision (input tokens measured (the system's own count))",
        "p50_s": 0.26079703494906425,
        "p95_s": 0.2831697516143322,
        "adjusted_p50_s": 0.6715940698981285,
        "adjusted_p95_s": 0.7163395032286645,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our RunPod GPU (H100 NVL 96 GB, Canada), reached over the internet from Germany",
        "repo": "https://huggingface.co/Mapika/decider-2b",
        "note": "The author's TypeSafe-compatible server and published weights (Qwen3.5-2B-Base with a trained one-pass decision readout), run serially on our GPU. Self-host latency gets the standard ×2 + 0.15 s adjustment."
      },
      {
        "key": "nimble-9b",
        "name": "Bespoke Nimble 9B (Bespoke Labs)",
        "rank": 24,
        "score": 60.47515137925396,
        "correct": 437,
        "n": 534,
        "accuracy_pct": 81.83520599250936,
        "cost_usd_per_1000": 0.16583014981273408,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, openrouter qwen/qwen3.5-9b list price $0.1/M in, $0.15/M out (a LoRA merge of Qwen3.5-9B; the base weights are listed on OpenRouter (size class dense_9B), as in the v1.1.3 row) x 970 input and 1 output tokens per decision (input tokens measured (the system's own count))",
        "p50_s": 0.3889440931379795,
        "p95_s": 0.6545137587934732,
        "adjusted_p50_s": 0.927888186275959,
        "adjusted_p95_s": 1.4590275175869463,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our RunPod GPU (A40 48 GB, Canada), reached over the internet from Germany",
        "repo": "https://github.com/bespokelabsai/nimble",
        "note": "Re-run in v1.2.8 at Bespoke Labs' request after they raised the serving prompt limit from 2,048 to 8,192 tokens (bespokelabsai/nimble PR #4). Same recipe as the v1.1.3 run — the published LoRA merged into Qwen3.5-9B with the author's PEFT safe-merge, served with SGLang and the author's Jev-compatible API — now from current nimble main; the adapter weights are unchanged. Hard-tier accuracy rose from 43.6 % to 65.5 %, yet the score fell: the long hard items that used to fail at once are now answered and priced (so Cost fell), and this pod was in Canada while the v1.1.3 run's was in Sweden, so part of the lower Speed is network distance from our server in Germany. This complete run replaces the earlier row; its old score is kept in the artifact under superseded_rows."
      },
      {
        "key": "gemini-3.1-flash-lite",
        "name": "Gemini 3.1 Flash-Lite",
        "rank": 25,
        "score": 60.08928259753951,
        "correct": 468,
        "n": 534,
        "accuracy_pct": 87.64044943820225,
        "cost_usd_per_1000": 0.26378698501872655,
        "cost_kind": "measured",
        "cost_basis": "public tariff x measured tokens (https://ai.google.dev/gemini-api/docs/pricing (paid tier, read 2026-09-19)) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | public tariff x measured tokens (hard-tier run)",
        "p50_s": 0.756230715662241,
        "p95_s": 0.8761593606323003,
        "adjusted_p50_s": 0.756230715662241,
        "adjusted_p95_s": 0.8761593606323003,
        "latency_adjustment": "none (production API)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "production API (Google)",
        "repo": null,
        "note": ""
      },
      {
        "key": "openjev-thinking",
        "name": "OpenJev (thinking, BF16)",
        "rank": 26,
        "score": 59.98618116922599,
        "correct": 478,
        "n": 534,
        "accuracy_pct": 89.51310861423221,
        "cost_usd_per_1000": 0.25463898876404495,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: same hosted reference x 1778 billed input and 315 thought output tokens per decision",
        "p50_s": 0.46296589844860137,
        "p95_s": 1.0775369299459272,
        "adjusted_p50_s": 1.0759317968972026,
        "adjusted_p95_s": 2.3050738598918543,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our GPU (lium.io H200 141 GB), reached over the internet from Germany; serial, one request at a time",
        "repo": "https://github.com/razorback16/openjev",
        "note": "OpenJev's real typed-API thinking switch at think=512, using its own /v1/systemone server over BF16 DiffusionGemma. The thought is generated first, then native probability reads are taken after it. All 534 requests returned valid distributions. Cost counts the server's billed input and thought output tokens."
      },
      {
        "key": "kev-4b",
        "name": "kev 4B (research preview)",
        "rank": 27,
        "score": 59.724673285425375,
        "correct": 378,
        "n": 534,
        "accuracy_pct": 70.78651685393258,
        "cost_usd_per_1000": 0.018802865168539323,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, DeepInfra Qwen3.5-4B size-class reference list price $0.03/M in, $0.0/M out (a 4B one-pass model with no generated output) x 279 input and 0 output tokens per decision (input tokens measured (the system's own count))",
        "p50_s": 0.5502474755048752,
        "p95_s": 0.9917014226317405,
        "adjusted_p50_s": 1.2504949510097503,
        "adjusted_p95_s": 2.1334028452634812,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our RunPod GPU (GeForce RTX 3090 24 GB, community cloud CA), reached over the internet",
        "repo": "https://github.com/jaredpalmer/kev",
        "note": "Self-hosted from the author's repository at commit 20fa626 through its native TypeSafe-compatible `/v1/systemone` server, BF16 on an RTX 3090; measured serially from Sandy over the internet. The author labels this checkpoint a research preview."
      },
      {
        "key": "deepseek-flash",
        "name": "DeepSeek V4.1 Flash (thinking default)",
        "rank": 28,
        "score": 57.54288064128547,
        "correct": 511,
        "n": 534,
        "accuracy_pct": 95.69288389513109,
        "cost_usd_per_1000": 0.593682584269663,
        "cost_kind": "measured",
        "cost_basis": "public tariff x measured tokens (https://api-docs.deepseek.com/quick_start/pricing (cache-miss off-peak; the run is on a Saturday, off-peak all day)) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | public tariff x measured tokens (hard-tier run)",
        "p50_s": 1.4163860343396664,
        "p95_s": 4.886470635980367,
        "adjusted_p50_s": 1.4163860343396664,
        "adjusted_p95_s": 4.886470635980367,
        "latency_adjustment": "none (production API)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "production API (DeepSeek)",
        "repo": null,
        "note": ""
      },
      {
        "key": "kev-8b",
        "name": "kev 8B (research preview)",
        "rank": 29,
        "score": 56.383551097366656,
        "correct": 397,
        "n": 534,
        "accuracy_pct": 74.34456928838952,
        "cost_usd_per_1000": 0.07333117415730338,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, OpenRouter qwen/qwen3-8b list price list price $0.117/M in, $0.0/M out (the same-size Qwen3-8B weights; kev generates no output tokens) x 279 input and 0 output tokens per decision (input tokens measured (the system's own count))",
        "p50_s": 0.5904169715940952,
        "p95_s": 1.1521932914853092,
        "adjusted_p50_s": 1.3308339431881904,
        "adjusted_p95_s": 2.4543865829706184,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our RunPod GPU (GeForce RTX 3090 24 GB, community cloud CA), reached over the internet",
        "repo": "https://github.com/jaredpalmer/kev",
        "note": "Self-hosted from the author's repository at commit 20fa626 through its native TypeSafe-compatible `/v1/systemone` server, BF16 on an RTX 3090; measured serially from Sandy over the internet. The author labels this checkpoint a research preview."
      },
      {
        "key": "open-jev-zefan-9b",
        "name": "Open-Jev 9B (Zefan Cai)",
        "rank": 30,
        "score": 54.95864804501749,
        "correct": 412,
        "n": 534,
        "accuracy_pct": 77.15355805243446,
        "cost_usd_per_1000": 0.24882153558052428,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, OpenRouter Qwen3.5-9B list price read 2026-09-21 list price $0.1/M in, $0.0/M out (the exact 9B base and a conservative same-family proxy for the unlisted 2B; the decision head generates no output tokens) x 1439 input and 0 output tokens per decision (input tokens measured (the system's own count))",
        "p50_s": 0.754859171807766,
        "p95_s": 1.8114654466509819,
        "adjusted_p50_s": 1.6597183436155318,
        "adjusted_p95_s": 3.7729308933019636,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our RunPod GPU (H100 80GB HBM3), reached over the internet",
        "repo": "https://github.com/Zefan-Cai/Open-Jev",
        "note": "The author's pinned LoRA adapter, trained scalar decision head and calibration temperature, served by the author's Open-Jev server with prefix caching off, batch size 1 and 4,096-token limit. Serial requests were measured from Sandy over an SSH tunnel to the H100. Self-host latency receives the standing x2 + 0.15 s adjustment. Cost uses the exact Qwen3.5-9B hosted input tariff for 9B and the same conservative same-family proxy for the unlisted 2B; neither receives an automatic 100. Exact normalized comparison found no JevBench public task state or instruction in the 79,116-row public training projection."
      },
      {
        "key": "system-one-sg",
        "name": "system-one (Qwen3-8B, Sean Goedecke)",
        "rank": 31,
        "score": 54.830990505255514,
        "correct": 403,
        "n": 534,
        "accuracy_pct": 75.46816479400749,
        "cost_usd_per_1000": 0.08944116853932584,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, openrouter qwen/qwen3-8b list price $0.117/M in, $0.455/M out (same weights, listed on OpenRouter) x 412 input and 1 output tokens per decision (input tokens measured) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | ESTIMATE: openrouter qwen/qwen3-8b $0.117/M in, $0.455/M out x 1258 in / 1 out tokens per hard decision",
        "p50_s": 0.1661309413611889,
        "p95_s": 0.30472867079079147,
        "adjusted_p50_s": 0.4822618827223778,
        "adjusted_p95_s": 0.759457341581583,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": "RunPod RTX PRO 4500 Blackwell 32 GB (EU-RO-1)",
        "measured_where": "from a Hetzner server in Germany over the internet to the pod's public TCP port (plain HTTP, one connection per request) through a thin transport around the author's library; model loaded before timing",
        "endpoint": "our RunPod GPU (RTX PRO 4500 Blackwell 32 GB (EU-RO-1)), reached over the internet",
        "repo": "https://github.com/sgoedecke/system-one",
        "note": ""
      },
      {
        "key": "jeff",
        "name": "jeff (Logan Markewich, GLiFormer 400M)",
        "rank": 32,
        "score": 54.38199454750733,
        "correct": 318,
        "n": 534,
        "accuracy_pct": 59.55056179775281,
        "cost_usd_per_1000": 0.006036722846441948,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, deepinfra encoders of the same size (bge-large, e5-large, Qwen3-Embedding-0.6B) list price $0.01/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 272 input and 0 output tokens per decision (input tokens measured (the system's own count))",
        "p50_s": 0.9379300177097321,
        "p95_s": 10.969045254960655,
        "adjusted_p50_s": 2.025860035419464,
        "adjusted_p95_s": 22.088090509921308,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads",
        "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing",
        "endpoint": "our CPU (4 threads, Ryzen 5 3600)",
        "repo": "https://github.com/logan-markewich/jeff",
        "note": "Self-hosted from its GitHub repo with server defaults, on our CPU (the author recommends a GPU, e.g. an L4), through the same TypeSafe-compatible API as Jev."
      },
      {
        "key": "laya",
        "name": "Laya (Convai Innovations, ModernBERT-large 421M)",
        "rank": 33,
        "score": 54.35373809070198,
        "correct": 314,
        "n": 534,
        "accuracy_pct": 58.80149812734082,
        "cost_usd_per_1000": 0.0028831273408239703,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, deepinfra encoders of the same size (bge-large, e5-large, Qwen3-Embedding-0.6B) list price $0.01/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 205 input and 0 output tokens per decision (input tokens measured (the system's own count))",
        "p50_s": 0.787067785859108,
        "p95_s": 2.197125389799475,
        "adjusted_p50_s": 1.7241355717182159,
        "adjusted_p95_s": 4.544250779598951,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads",
        "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing",
        "endpoint": "our CPU (4 threads, Ryzen 5 3600)",
        "repo": "https://huggingface.co/convaiinnovations/laya",
        "note": "The English checkpoint (repo root), run on our CPU through its own `laya` package. Its budget is 512 tokens per question, so long hard-tier states are cut by the package itself."
      },
      {
        "key": "open-jev-zefan-2b",
        "name": "Open-Jev 2B (Zefan Cai)",
        "rank": 34,
        "score": 51.31393833805917,
        "correct": 371,
        "n": 534,
        "accuracy_pct": 69.47565543071161,
        "cost_usd_per_1000": 0.24882153558052428,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, OpenRouter Qwen3.5-9B list price read 2026-09-21 list price $0.1/M in, $0.0/M out (the exact 9B base and a conservative same-family proxy for the unlisted 2B; the decision head generates no output tokens) x 1439 input and 0 output tokens per decision (input tokens measured (the system's own count))",
        "p50_s": 0.664745207875967,
        "p95_s": 1.450564834475517,
        "adjusted_p50_s": 1.479490415751934,
        "adjusted_p95_s": 3.051129668951034,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our RunPod GPU (H100 80GB HBM3), reached over the internet",
        "repo": "https://github.com/Zefan-Cai/Open-Jev",
        "note": "The author's pinned LoRA adapter, trained scalar decision head and calibration temperature, served by the author's Open-Jev server with prefix caching off, batch size 1 and 4,096-token limit. Serial requests were measured from Sandy over an SSH tunnel to the H100. Self-host latency receives the standing x2 + 0.15 s adjustment. Cost uses the exact Qwen3.5-9B hosted input tariff for 9B and the same conservative same-family proxy for the unlisted 2B; neither receives an automatic 100. Exact normalized comparison found no JevBench public task state or instruction in the 79,116-row public training projection."
      },
      {
        "key": "opendecision",
        "name": "OpenDecision (ModernBERT-large zero-shot)",
        "rank": 35,
        "score": 40.58744897671299,
        "correct": 300,
        "n": 534,
        "accuracy_pct": 56.17977528089888,
        "cost_usd_per_1000": 0.006635318352059925,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, deepinfra encoders of the same size (bge-large, e5-large, Qwen3-Embedding-0.6B) list price $0.01/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 329 input and 0 output tokens per decision (input tokens measured (the system's own count))",
        "p50_s": 0.3378134034574032,
        "p95_s": 0.5447697393596171,
        "adjusted_p50_s": 0.8256268069148064,
        "adjusted_p95_s": 1.2395394787192342,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our RunPod GPU (H100 NVL 96 GB, Canada), reached over the internet from Germany",
        "repo": "https://github.com/deepanwadhwa/OpenDecision",
        "note": "A zero-shot NLI classifier behind a TypeSafe-compatible server, not a trained decision model: it scores each option as an entailment hypothesis with ModernBERT-large-zeroshot-v2.0. Its choice path runs several NLI passes over the same state, which the reported token count does not include, so a per-token hosted price would be higher than the estimate here. Pre-registered for our CPU in v1.2.7, run on our GPU because the CPU was far too slow."
      },
      {
        "key": "openjev-verdict-1.4",
        "name": "openJev Verdict 1.4",
        "rank": 36,
        "score": 38.93683519894202,
        "correct": 292,
        "n": 534,
        "accuracy_pct": 54.68164794007491,
        "cost_usd_per_1000": 0.003872921348314607,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, deepinfra base-size encoders (bge-base, e5-base, gte-base, all-mpnet-base) list price $0.005/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 452 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts)",
        "p50_s": 0.31351133808493614,
        "p95_s": 0.9248626325279473,
        "adjusted_p50_s": 0.7770226761698723,
        "adjusted_p95_s": 1.9997252650558945,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads",
        "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing",
        "endpoint": "our CPU (4 threads, Ryzen 5 3600)",
        "repo": "https://huggingface.co/heman10x/rlcd-modernbert-151m",
        "note": "Same public weights as the earlier Verdict row, run through the author's fixed v1.4 engine. That engine auto-loads the calibrator for every option count, frames candidate labels as NLI sentences and uses a 512-token context budget. Run locally on our CPU, serially."
      },
      {
        "key": "openjev-verdict",
        "name": "openJev Verdict (heman10x, ModernBERT-base 151M)",
        "rank": 37,
        "score": 38.059545049974176,
        "correct": 298,
        "n": 534,
        "accuracy_pct": 55.80524344569289,
        "cost_usd_per_1000": 0.00367125468164794,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, deepinfra base-size encoders (bge-base, e5-base, gte-base, all-mpnet-base) list price $0.005/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 383 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json]",
        "p50_s": 0.27805980294942856,
        "p95_s": 1.4451411496847864,
        "adjusted_p50_s": 0.7061196058988571,
        "adjusted_p95_s": 3.0402822993695726,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads",
        "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing",
        "endpoint": "our CPU (4 threads, Ryzen 5 3600)",
        "repo": "https://github.com/Heman10x-NGU/openJev-verdict-2.0",
        "note": "The openJev-verdict-2.0 Hugging Face repo ships no weights; its config is byte-identical to heman10x/rlcd-modernbert-151m, whose published weights we ran with the author's engine. The 'verdict2-base' checkpoint behind the README's numbers is not downloadable yet (Git LFS 404); we will run it once it is."
      },
      {
        "key": "kev-0.5b",
        "name": "kev 0.5B",
        "rank": 38,
        "score": 33.24350332804402,
        "correct": 291,
        "n": 534,
        "accuracy_pct": 54.49438202247191,
        "cost_usd_per_1000": 0.0062676404494382025,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, DeepInfra Qwen3-Embedding-0.6B size-class reference list price $0.01/M in, $0.0/M out (a <=0.6B one-pass model with no generated output) x 279 input and 0 output tokens per decision (input tokens measured (the system's own count))",
        "p50_s": 0.43036095052957535,
        "p95_s": 0.9219581566751004,
        "adjusted_p50_s": 1.0107219010591506,
        "adjusted_p95_s": 1.9939163133502007,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our RunPod GPU (GeForce RTX 3090 24 GB, community cloud CA), reached over the internet",
        "repo": "https://github.com/jaredpalmer/kev",
        "note": "Self-hosted from the author's repository at commit 20fa626 through its native TypeSafe-compatible `/v1/systemone` server, BF16 on an RTX 3090; measured serially from Sandy over the internet. This is the v0.1 release."
      },
      {
        "key": "gliner2-large",
        "name": "GLiNER2 large (Fastino)",
        "rank": 39,
        "score": 29.55158409895235,
        "correct": 300,
        "n": 534,
        "accuracy_pct": 56.17977528089888,
        "cost_usd_per_1000": 0.007745842696629214,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, deepinfra encoders of the same size (bge-large, e5-large, Qwen3-Embedding-0.6B) list price $0.01/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 452 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts)",
        "p50_s": 1.0967330671846867,
        "p95_s": 14.48837994225323,
        "adjusted_p50_s": 2.3434661343693732,
        "adjusted_p95_s": 29.12675988450646,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads",
        "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing",
        "endpoint": "our CPU (4 threads, Ryzen 5 3600)",
        "repo": "https://huggingface.co/fastino/gliner2-large-v1",
        "note": "The large checkpoint of Fastino's earlier GLiNER2 family, same documented mapping as the GLiNER2 row: the question goes in front of the text and the probabilities are the model's own single-label softmax over the labels, read out in full. A general schema classifier, not a Jev rebuild."
      },
      {
        "key": "smalljev",
        "name": "smalljev semantic-v9",
        "rank": 40,
        "score": 27.444453253651357,
        "correct": 279,
        "n": 534,
        "accuracy_pct": 52.24719101123596,
        "cost_usd_per_1000": 0.025365692883895133,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, submitted Qwen/Qwen2.5-3B-Instruct hosted reference list price $0.04/M in, $0.0/M out (the author's documented reference for the same approximate size class; one forward pass, nothing generated) x 329 input and 0 output tokens per decision (input tokens measured (the system's own count))",
        "p50_s": 0.4138255603611469,
        "p95_s": 0.45828209072351456,
        "adjusted_p50_s": 0.9776511207222939,
        "adjusted_p95_s": 1.066564181447029,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our GPU (lium.io A6000 48 GB), reached over the internet from Germany; serial, one request at a time",
        "repo": "https://github.com/isHeSatoshi/smalljev",
        "note": "The public semantic-v9 LoRA and native heads over MiniCPM5-2B-Base, through the mapping frozen before the run. It has a typed Python contract but no TypeSafe-compatible HTTP route. The released training recipe explicitly hill-climbed against JevBench's public shape and source families; this allowed public benchmark-directed development is disclosed. Cost is $0.04/M measured input tokens, not free/100."
      },
      {
        "key": "gliner2",
        "name": "GLiNER2 (Fastino, gliner2.5-base)",
        "rank": 41,
        "score": 24.036445808742133,
        "correct": 281,
        "n": 534,
        "accuracy_pct": 52.62172284644194,
        "cost_usd_per_1000": 0.00367125468164794,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, deepinfra base-size encoders (bge-base, e5-base, gte-base, all-mpnet-base) list price $0.005/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 383 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json]",
        "p50_s": 0.31302378326654434,
        "p95_s": 4.153845678269863,
        "adjusted_p50_s": 0.7760475665330887,
        "adjusted_p95_s": 8.457691356539726,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads",
        "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing",
        "endpoint": "our CPU (4 threads, Ryzen 5 3600)",
        "repo": "https://github.com/fastino-ai/GLiNER2",
        "note": "A general schema classifier, not a Jev rebuild. The question goes in front of the text; the probabilities are GLiNER2's own single-label softmax over the labels, read out in full (mapping fixed before the run)."
      },
      {
        "key": "open-jev-deberta-v3-large",
        "name": "open-jev-deberta-v3-large (local CPU)",
        "rank": 42,
        "score": 23.065925583996304,
        "correct": 277,
        "n": 534,
        "accuracy_pct": 51.87265917602997,
        "cost_usd_per_1000": 0.007340468164794008,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, deepinfra encoders of the same size (bge-large, e5-large, Qwen3-Embedding-0.6B) list price $0.01/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 383 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | ESTIMATE: deepinfra encoders of the same size (bge-large, e5-large, Qwen3-Embedding-0.6B) $0.01/M in, $0.0/M out x 1235 in / 0 out tokens per hard decision",
        "p50_s": 1.767673410475254,
        "p95_s": 3.349288306012749,
        "adjusted_p50_s": 3.685346820950508,
        "adjusted_p95_s": 6.848576612025498,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "local, 2 CPU threads of a Ryzen 5 3600",
        "endpoint": "our CPU (2 threads, Ryzen 5 3600)",
        "repo": "https://github.com/kotoba-lang/typed-decisions",
        "note": ""
      },
      {
        "key": "gliner2.5-multi",
        "name": "GLiNER2.5 multi (Fastino, 287M)",
        "rank": 43,
        "score": 16.61305398280731,
        "correct": 261,
        "n": 534,
        "accuracy_pct": 48.87640449438202,
        "cost_usd_per_1000": 0.003872921348314607,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, deepinfra base-size encoders (bge-base, e5-base, gte-base, all-mpnet-base) list price $0.005/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 452 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts)",
        "p50_s": 0.42792757973074913,
        "p95_s": 8.17526703067124,
        "adjusted_p50_s": 1.0058551594614982,
        "adjusted_p95_s": 16.500534061342478,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads",
        "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing",
        "endpoint": "our CPU (4 threads, Ryzen 5 3600)",
        "repo": "https://huggingface.co/fastino/gliner2.5-multi-v1",
        "note": "The multilingual GLiNER2.5 checkpoint (287M), same family and same documented mapping as the GLiNER2 row. JevBench items are English only, so its multilingual training is not exercised here."
      },
      {
        "key": "gliner2.5-small",
        "name": "GLiNER2.5 small (Fastino, 74M)",
        "rank": 44,
        "score": 13.84974486725046,
        "correct": 252,
        "n": 534,
        "accuracy_pct": 47.19101123595505,
        "cost_usd_per_1000": 0.003872921348314607,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, deepinfra base-size encoders (bge-base, e5-base, gte-base, all-mpnet-base) list price $0.005/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 452 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts)",
        "p50_s": 0.11413825303316116,
        "p95_s": 2.1012388937175266,
        "adjusted_p50_s": 0.37827650606632235,
        "adjusted_p95_s": 4.3524777874350535,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads",
        "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing",
        "endpoint": "our CPU (4 threads, Ryzen 5 3600)",
        "repo": "https://huggingface.co/fastino/gliner2.5-small-v1",
        "note": "The small GLiNER2.5 checkpoint (74M), same family and same documented mapping as the GLiNER2 row: the question goes in front of the text and the probabilities are the model's own single-label softmax over the labels, read out in full. A general schema classifier, not a Jev rebuild."
      },
      {
        "key": "mxbai-rerank-base-v2",
        "name": "Mixedbread mxbai-rerank-base-v2",
        "rank": 45,
        "score": 0.764251199794221,
        "correct": 191,
        "n": 534,
        "accuracy_pct": 35.767790262172284,
        "cost_usd_per_1000": 0.011722048450570633,
        "cost_kind": "measured",
        "cost_basis": "MEASURED model inference time x lium.io A6000 tariff USD 0.42/hour",
        "p50_s": 0.06881788885220885,
        "p95_s": 0.23386348099447787,
        "adjusted_p50_s": 0.2876357777044177,
        "adjusted_p95_s": 0.6177269619889557,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our GPU (lium.io A6000 48 GB), serial, one option batch per decision",
        "repo": "https://huggingface.co/mixedbread-ai/mxbai-rerank-base-v2",
        "note": "Neutral documented reranker adapter; instruction and no-instruction public calibration were run, then frozen before one held-out pass."
      },
      {
        "key": "bge-reranker-v2-m3",
        "name": "BAAI bge-reranker-v2-m3",
        "rank": 46,
        "score": 0.6763072957019215,
        "correct": 160,
        "n": 534,
        "accuracy_pct": 29.962546816479403,
        "cost_usd_per_1000": 0.007718915951068509,
        "cost_kind": "measured",
        "cost_basis": "MEASURED model inference time x lium.io A6000 tariff USD 0.42/hour",
        "p50_s": 0.034560746513307095,
        "p95_s": 0.17954895906150325,
        "adjusted_p50_s": 0.21912149302661418,
        "adjusted_p95_s": 0.5090979181230065,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our GPU (lium.io A6000 48 GB), serial, one option batch per decision",
        "repo": "https://huggingface.co/BAAI/bge-reranker-v2-m3",
        "note": "Neutral documented reranker adapter; instruction and no-instruction public calibration were run, then frozen before one held-out pass."
      },
      {
        "key": "gte-reranker-modernbert-base",
        "name": "Alibaba GTE Reranker ModernBERT-base",
        "rank": 47,
        "score": 0.32219057759911496,
        "correct": 180,
        "n": 534,
        "accuracy_pct": 33.70786516853933,
        "cost_usd_per_1000": 0.010280148864935354,
        "cost_kind": "measured",
        "cost_basis": "MEASURED model inference time x lium.io A6000 tariff USD 0.42/hour",
        "p50_s": 0.048107756301760674,
        "p95_s": 0.10235980194993316,
        "adjusted_p50_s": 0.24621551260352134,
        "adjusted_p95_s": 0.35471960389986634,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our GPU (lium.io A6000 48 GB), serial, one option batch per decision",
        "repo": "https://huggingface.co/Alibaba-NLP/gte-reranker-modernbert-base",
        "note": "Neutral documented reranker adapter; instruction and no-instruction public calibration were run, then frozen before one held-out pass."
      },
      {
        "key": "certo",
        "name": "Certo v1 (AltSlate Labs)",
        "rank": 48,
        "score": 0.0,
        "correct": 151,
        "n": 534,
        "accuracy_pct": 28.277153558052436,
        "cost_usd_per_1000": 0.0009654868913857679,
        "cost_kind": "estimate",
        "cost_basis": "ESTIMATE: hosted-provider price, deepinfra encoders of the same size (bge-large, e5-large, Qwen3-Embedding-0.6B) list price $0.01/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 86 input and 0 output tokens per decision (input tokens measured (the system's own count))",
        "p50_s": 0.019205978140234947,
        "p95_s": 0.031265050335787234,
        "adjusted_p50_s": 0.1884119562804699,
        "adjusted_p95_s": 0.21253010067157446,
        "latency_adjustment": "x2 + 0.15 s (assumption, not measured)",
        "latency_run": "serial 242-decision standard+judge run",
        "hardware": null,
        "measured_where": "from a Hetzner server in Germany, network included",
        "endpoint": "our RunPod GPU (GeForce RTX 3090 24 GB, community cloud), reached over the internet from Germany",
        "repo": "https://huggingface.co/altslate/certo-decision-model",
        "note": "The public Certo v1 checkpoint through the author's DecisionModel, serially on our rented GPU. The question instruction is prepended to the state because Certo exposes state + runtime options but no separate question field; the published 64-token state and 48-token option limits are unchanged. The model card says v1 does not yet transfer to arbitrary natural-language prose. Cost is an estimate from same-size hosted encoders times the checkpoint's retained input tokens, not free/100."
      }
    ],
    "unranked": [
      {
        "key": "classifier-dev-fast",
        "name": "classifier.dev (fast tier)",
        "reason": "runs on Jev (TypeSafe) — listed, not ranked",
        "score": 83.64670446057299
      },
      {
        "key": "qwen3.8-27b",
        "name": "Qwen3.8 27B (Chutes TEE)",
        "reason": "Partial run; not ranked",
        "score": 24.836695615027768
      },
      {
        "key": "needle-3-tools",
        "name": "Needle 3, options as tools (post-hoc adapter mode)",
        "reason": "Partial run; not ranked",
        "score": 1.0757743846539873
      },
      {
        "key": "needle-3",
        "name": "Needle 3 (Cactus, 2-bit, local CPU)",
        "reason": "Partial run; not ranked",
        "score": 0.09466799356370086
      }
    ]
  },
  "creative": {
    "name": "Creative Judge v1",
    "questions": 100,
    "protocol_sha256": "8514f25341c8b60cc90ce62b799e238284050750e4b3557958415fae515206de",
    "rate_usd_per_gpu_second": 0.000542,
    "rate_source": "https://modal.com/pricing",
    "cost_method": "Serial: mean inference call time × GPU rate (L4, L40S or H100, as labeled). Batched: measured two-pass H100 throughput; first-pass overhead retained, no prefix cache. GPU-only before credits; CPU/RAM and startup excluded.",
    "warmup_caveat": "Base Laya began cold; typed-decisions reused a warm container. This affects their mean-cost ordering and observed frontier. No timings dropped.",
    "models": [
      {
        "key": "9B",
        "name": "Open-Jev 9B",
        "correct": 78,
        "n": 100,
        "accuracy_pct": 78,
        "mean_s": 0.5254166906500006,
        "p50_s": 0.6305679275000049,
        "p95_s": 0.8388852216500001,
        "timing_n": 100,
        "cost_usd_per_1000": 0.2847758463323003,
        "cost_kind": "gpu-proxy"
      },
      {
        "key": "2B",
        "name": "Open-Jev 2B",
        "correct": 62,
        "n": 100,
        "accuracy_pct": 62,
        "mean_s": 0.4265552508599999,
        "p50_s": 0.47494447800000117,
        "p95_s": 0.6382979529499996,
        "timing_n": 100,
        "cost_usd_per_1000": 0.23119294596611992,
        "cost_kind": "gpu-proxy"
      },
      {
        "key": "kev",
        "name": "Kev-0.8B",
        "correct": 58,
        "n": 100,
        "accuracy_pct": 58,
        "mean_s": 0.0909383176499999,
        "p50_s": 0.07819289050000044,
        "p95_s": 0.08913331305000086,
        "timing_n": 100,
        "cost_usd_per_1000": 0.049288568166299944,
        "cost_kind": "gpu-proxy"
      },
      {
        "key": "laya",
        "name": "Laya English base",
        "correct": 50,
        "n": 100,
        "accuracy_pct": 50,
        "mean_s": 0.044768103570000085,
        "p50_s": 0.03145998849999998,
        "p95_s": 0.044756079999999615,
        "timing_n": 100,
        "cost_usd_per_1000": 0.024264312134940045,
        "cost_kind": "gpu-proxy"
      },
      {
        "key": "laya-typed",
        "name": "Laya typed-decisions",
        "correct": 51,
        "n": 100,
        "accuracy_pct": 51,
        "mean_s": 0.03564822323999998,
        "p50_s": 0.03547657550000238,
        "p95_s": 0.04251481310000251,
        "timing_n": 100,
        "cost_usd_per_1000": 0.019321336996079987,
        "cost_kind": "gpu-proxy"
      },
      {
        "key": "semif",
        "name": "SemIf Qwen3.5-4B",
        "correct": 86,
        "n": 100,
        "accuracy_pct": 86,
        "mean_s": 0.06972967674000007,
        "p50_s": 0.06848917599999815,
        "p95_s": 0.08891518959999906,
        "timing_n": 100,
        "cost_usd_per_1000": 0.07650895086750008,
        "cost_kind": "gpu-proxy",
        "gpu_rate_usd_per_second": 0.0010972222222222223,
        "gpu": "H100",
        "batch_cost_usd_per_1000": 0.013331274852083323,
        "batch_size": 32,
        "throughput_dps": 82.30437331736175,
        "batch_agreement": 0.99,
        "source_aggregate": "creative-expansion.json"
      },
      {
        "key": "jevfire",
        "name": "Jevfire Qwen3.8-27B FP8",
        "correct": 94,
        "n": 100,
        "accuracy_pct": 94,
        "mean_s": 0.10931126653999924,
        "p50_s": 0.10787691150000711,
        "p95_s": 0.11089101779998883,
        "timing_n": 100,
        "cost_usd_per_1000": 0.11993875078694363,
        "cost_kind": "gpu-proxy",
        "gpu_rate_usd_per_second": 0.0010972222222222223,
        "gpu": "H100",
        "batch_cost_usd_per_1000": 0.04792116926513909,
        "batch_size": 16,
        "throughput_dps": 22.89639921245435,
        "batch_agreement": 0.97,
        "source_aggregate": "creative-expansion.json"
      },
      {
        "key": "diffusion",
        "name": "JoshuaSP DiffusionGemma 26B-A4B · 1 step",
        "correct": 96,
        "n": 100,
        "accuracy_pct": 96,
        "mean_s": 0.29053775290999906,
        "p50_s": 0.2442433664999939,
        "p95_s": 0.3329541904000044,
        "timing_n": 100,
        "cost_usd_per_1000": 0.3187844788873601,
        "cost_kind": "gpu-proxy",
        "gpu_rate_usd_per_second": 0.0010972222222222223,
        "gpu": "H100",
        "batch_cost_usd_per_1000": 0.04836777897215271,
        "batch_size": 8,
        "throughput_dps": 22.68498255530686,
        "batch_agreement": 0.98,
        "source_aggregate": "creative-expansion.json"
      },
      {
        "key": "autojev",
        "name": "AutoJev-27B",
        "correct": 99,
        "n": 100,
        "accuracy_pct": 99,
        "mean_s": 0.10365273339000068,
        "p50_s": 0.10129723800000079,
        "p95_s": 0.11264144520000983,
        "timing_n": 100,
        "cost_usd_per_1000": 0.11373008246958409,
        "cost_kind": "gpu-proxy",
        "gpu_rate_usd_per_second": 0.0010972222222222223,
        "gpu": "H100",
        "batch_cost_usd_per_1000": 0.03375328504395823,
        "batch_size": 4,
        "throughput_dps": 32.50712399676851,
        "batch_agreement": 1.0,
        "source_aggregate": "creative-autojev.json"
      },
      {
        "key": "jevk5",
        "name": "JevK5 v0.2 · L4",
        "correct": 89,
        "n": 100,
        "accuracy_pct": 89,
        "mean_s": 0.06022846326000149,
        "p50_s": 0.05808848149999335,
        "p95_s": 0.06875913165001464,
        "timing_n": 100,
        "cost_usd_per_1000": 0.013384102946666998,
        "cost_kind": "gpu-proxy",
        "gpu_rate_usd_per_second": 0.00022222222222222223,
        "gpu": "L4",
        "batch_cost_usd_per_1000": null,
        "batch_size": null,
        "throughput_dps": null,
        "batch_agreement": null,
        "source_aggregate": "jevk5-reproduction.json"
      }
    ],
    "expansion_source": "creative-expansion.json",
    "latest_run_date": "2026-09-23",
    "autojev_source": "creative-autojev.json",
    "jevk5_source": "jevk5-reproduction.json"
  },
  "decision_index": {
    "edition": "Decision Index 0.1",
    "commit": "944cdb0b6b3ba9bc51905ed26b100b5a041d22ff",
    "snapshot_date": "2026-09-22",
    "generated_utc": "2026-09-22T11:12:47+00:00",
    "source_url": "https://huggingface.co/spaces/multimodalart/jev-decision-index/blob/944cdb0b6b3ba9bc51905ed26b100b5a041d22ff/data/index.json",
    "source_sha256": "4531c40efcd35986ac6b93d05b646c8a9e496c3833c20e6a9d748c9dba4688f1",
    "methodology_url": "https://huggingface.co/spaces/multimodalart/jev-decision-index/blob/944cdb0b6b3ba9bc51905ed26b100b5a041d22ff/data/methodology.json",
    "methodology_sha256": "71f294c654992ceb1ca74e848d731a14a8c94480f919ba5d27aca40538922680",
    "score_field": "scores.balanced_raw",
    "score_method": "Published Decision Index: mean of five equally weighted category scores, each a mean of its panel benchmarks. Nineteen benchmarks; six interactive environments excluded for every model. Not percent correct.",
    "planned_requests": 132422,
    "scoreable_requests": 131980,
    "cost_usd_per_1000": null,
    "cost_note": "No comparable per-model cost series in the source; no cost frontier computed.",
    "models": [
      {
        "key": "jev",
        "name": "Jev",
        "score": 59.51,
        "p50_s": 0.2528,
        "p95_s": 0.43660000000000004,
        "timing_kind": "remote-api",
        "timing_note": "HTTPS request round trip; network and shared service scheduling included."
      },
      {
        "key": "jevfire-uncapped",
        "name": "Jevfire",
        "score": 55.74,
        "p50_s": 0.0825,
        "p95_s": 1.5629000000000002,
        "timing_kind": "loopback",
        "timing_note": "Local HTTP loopback; native serving and request overhead included."
      },
      {
        "key": "joshua-diffusion-full",
        "name": "JoshuaSP diffusiongemma (open-jev)",
        "score": 55.56,
        "p50_s": 0.2691,
        "p95_s": 1.6809,
        "timing_kind": "in-process",
        "timing_note": "Synchronized local GPU request time, including prompt preparation; excludes internet transit and model loading."
      },
      {
        "key": "decider-35b-nvfp4",
        "name": "Decider 35B-A3B",
        "score": 54.34,
        "p50_s": 0.0993,
        "p95_s": 0.3665,
        "timing_kind": "in-process",
        "timing_note": "Synchronized local GPU request time, including prompt preparation; excludes internet transit and model loading."
      },
      {
        "key": "razorback-one-read",
        "name": "mmastrac diffusiongemma vLLM",
        "score": 51.52,
        "p50_s": 0.040600000000000004,
        "p95_s": 0.3408,
        "timing_kind": "loopback",
        "timing_note": "Local HTTP loopback; native serving and request overhead included."
      },
      {
        "key": "kev-9b-raised",
        "name": "Kev 9B",
        "score": 50.48,
        "p50_s": 0.0541,
        "p95_s": 0.9482,
        "timing_kind": "in-process",
        "timing_note": "Synchronized local GPU request time, including prompt preparation; excludes internet transit and model loading."
      },
      {
        "key": "solomon-v11-bf16-encoding",
        "name": "Solomon v1.1",
        "score": 47.51,
        "p50_s": 0.22319999999999998,
        "p95_s": 5.301,
        "timing_kind": "unrecorded",
        "timing_note": "The source publishes timings but does not record the measurement method."
      },
      {
        "key": "kev-4b-raised",
        "name": "Kev 4B",
        "score": 47.43,
        "p50_s": 0.0538,
        "p95_s": 0.6271,
        "timing_kind": "in-process",
        "timing_note": "Synchronized local GPU request time, including prompt preparation; excludes internet transit and model loading."
      },
      {
        "key": "kev-8b-raised",
        "name": "Kev 8B",
        "score": 46.55,
        "p50_s": 0.038299999999999994,
        "p95_s": 0.6527000000000001,
        "timing_kind": "in-process",
        "timing_note": "Synchronized local GPU request time, including prompt preparation; excludes internet transit and model loading."
      },
      {
        "key": "pngwn-space-uncapped-full",
        "name": "open-jev (pngwn)",
        "score": 46.15,
        "p50_s": 0.1385,
        "p95_s": 2.0793000000000004,
        "timing_kind": "in-process",
        "timing_note": "Synchronized local GPU request time, including prompt preparation; excludes internet transit and model loading."
      },
      {
        "key": "openvons",
        "name": "openvons",
        "score": 45.59,
        "p50_s": 0.018,
        "p95_s": 0.2959,
        "timing_kind": "loopback",
        "timing_note": "Local HTTP loopback; native serving and request overhead included."
      },
      {
        "key": "decision-nox-full",
        "name": "Decision 1.0 Nox",
        "score": 45.31,
        "p50_s": 0.048799999999999996,
        "p95_s": 0.6195,
        "timing_kind": "unrecorded",
        "timing_note": "The source publishes timings but does not record the measurement method."
      },
      {
        "key": "semif",
        "name": "SemIf",
        "score": 44.77,
        "p50_s": 0.11040000000000001,
        "p95_s": 0.9586,
        "timing_kind": "in-process",
        "timing_note": "Synchronized local GPU request time, including prompt preparation; excludes internet transit and model loading."
      },
      {
        "key": "decider-2b-fp8-http",
        "name": "Decider 2B",
        "score": 44.0,
        "p50_s": 0.0497,
        "p95_s": 1.6409,
        "timing_kind": "loopback",
        "timing_note": "Local HTTP loopback; native serving and request overhead included. Concurrency 8; includes continuous-batcher queueing."
      },
      {
        "key": "mini-jev",
        "name": "mini-jev",
        "score": 43.48,
        "p50_s": 0.0645,
        "p95_s": 0.8342999999999999,
        "timing_kind": "in-process",
        "timing_note": "Synchronized local GPU request time, including prompt preparation; excludes internet transit and model loading."
      },
      {
        "key": "decision-sol",
        "name": "Decision 1.0 Sol",
        "score": 40.41,
        "p50_s": 0.0378,
        "p95_s": 0.31560000000000005,
        "timing_kind": "unrecorded",
        "timing_note": "The source publishes timings but does not record the measurement method."
      },
      {
        "key": "nimble",
        "name": "Bespoke Nimble 9B",
        "score": 34.21,
        "p50_s": 0.0835,
        "p95_s": 0.1369,
        "timing_kind": "in-process",
        "timing_note": "Synchronized local GPU request time, including prompt preparation; excludes internet transit and model loading."
      },
      {
        "key": "kev-0.6b-raised",
        "name": "Kev 0.6B",
        "score": 31.3,
        "p50_s": 0.0228,
        "p95_s": 0.1809,
        "timing_kind": "in-process",
        "timing_note": "Synchronized local GPU request time, including prompt preparation; excludes internet transit and model loading."
      },
      {
        "key": "kev-0.5b-raised",
        "name": "Kev 0.5B",
        "score": 30.34,
        "p50_s": 0.016800000000000002,
        "p95_s": 0.0856,
        "timing_kind": "in-process",
        "timing_note": "Synchronized local GPU request time, including prompt preparation; excludes internet transit and model loading."
      },
      {
        "key": "harsha",
        "name": "Qwen-2.5-1B-RLCD",
        "score": 28.81,
        "p50_s": 0.0424,
        "p95_s": 0.2949,
        "timing_kind": "in-process",
        "timing_note": "Synchronized local GPU request time, including prompt preparation; excludes internet transit and model loading."
      },
      {
        "key": "lfm2600",
        "name": "LFM2.5-2.6B-RLCD",
        "score": 27.27,
        "p50_s": 0.038700000000000005,
        "p95_s": 0.09409999999999999,
        "timing_kind": "in-process",
        "timing_note": "Synchronized local GPU request time, including prompt preparation; excludes internet transit and model loading."
      },
      {
        "key": "jeff-uncapped",
        "name": "jeff",
        "score": 27.23,
        "p50_s": 0.0224,
        "p95_s": 0.1579,
        "timing_kind": "in-process",
        "timing_note": "Synchronized local GPU request time, including prompt preparation; excludes internet transit and model loading."
      },
      {
        "key": "nanojev-8k",
        "name": "NanoJev",
        "score": 26.19,
        "p50_s": 0.027600000000000003,
        "p95_s": 0.9557,
        "timing_kind": "in-process",
        "timing_note": "Synchronized local GPU request time, including prompt preparation; excludes internet transit and model loading."
      },
      {
        "key": "lfm350-sdpa",
        "name": "LFM2.5-350M-RLCD",
        "score": 25.79,
        "p50_s": 0.0264,
        "p95_s": 0.37860000000000005,
        "timing_kind": "in-process",
        "timing_note": "Synchronized local GPU request time, including prompt preparation; excludes internet transit and model loading."
      },
      {
        "key": "gliner-base",
        "name": "GLiNER 2.5 base",
        "score": 24.7,
        "p50_s": 0.0144,
        "p95_s": 0.0682,
        "timing_kind": "in-process",
        "timing_note": "Synchronized local GPU request time, including prompt preparation; excludes internet transit and model loading."
      },
      {
        "key": "gliner-small",
        "name": "GLiNER 2.5 small",
        "score": 23.93,
        "p50_s": 0.0145,
        "p95_s": 0.0384,
        "timing_kind": "in-process",
        "timing_note": "Synchronized local GPU request time, including prompt preparation; excludes internet transit and model loading."
      },
      {
        "key": "gliner-multi",
        "name": "GLiNER 2.5 multilingual",
        "score": 22.42,
        "p50_s": 0.0152,
        "p95_s": 0.0831,
        "timing_kind": "in-process",
        "timing_note": "Synchronized local GPU request time, including prompt preparation; excludes internet transit and model loading."
      },
      {
        "key": "decision-lex",
        "name": "Decision 1.0 Lex",
        "score": 19.57,
        "p50_s": 0.0245,
        "p95_s": 0.1389,
        "timing_kind": "unrecorded",
        "timing_note": "The source publishes timings but does not record the measurement method."
      },
      {
        "key": "decision-kai",
        "name": "Decision 1.0 Kai",
        "score": 18.37,
        "p50_s": 0.0247,
        "p95_s": 0.1393,
        "timing_kind": "unrecorded",
        "timing_note": "The source publishes timings but does not record the measurement method."
      },
      {
        "key": "akash-gemma",
        "name": "system-one-gemma",
        "score": 17.09,
        "p50_s": 0.032100000000000004,
        "p95_s": 1.2728,
        "timing_kind": "in-process",
        "timing_note": "Synchronized local GPU request time, including prompt preparation; excludes internet transit and model loading."
      },
      {
        "key": "laya",
        "name": "Laya",
        "score": 16.39,
        "p50_s": 0.0189,
        "p95_s": 0.07479999999999999,
        "timing_kind": "in-process",
        "timing_note": "Synchronized local GPU request time, including prompt preparation; excludes internet transit and model loading."
      },
      {
        "key": "verdict",
        "name": "Verdict",
        "score": 13.38,
        "p50_s": 0.011800000000000001,
        "p95_s": 0.1678,
        "timing_kind": "in-process",
        "timing_note": "Synchronized local GPU request time, including prompt preparation; excludes internet transit and model loading."
      }
    ]
  }
}
