{
  "status": "frozen_before_inference",
  "date": "2026-09-21",
  "tasks": 231,
  "dataset_sha256": "dc3995d8ae1e2fc8e81ce38431add509eb8bb39b85aadfd0c7c32079382dde51",
  "jevbench_commit": "75e6224ed8103bbc3485ca74820a2eaf7ce8abe0",
  "measurement": "One serial stream of all 231 public tasks per model; no warmups, retries, answer selection or reordering. Native API probabilities scored, rounded as upstream returns them. State/question payloads and candidate order match our Open-Jev runs. Synchronized GPU-process timing excludes startup, network/RPC and the separate input audit; prediction failures retain elapsed time when inference started.",
  "scoring": "Unmodified upstream score_task: exact labels, sum tolerance 0.001, renormalize within 0.02, lexical argmax ties; failures and invalid probabilities count incorrect. Calibration uses valid distributions only. Ordinal accuracy is argmax; expected-value MAE separate.",
  "models": {
    "laya": {
      "model": "Laya English base",
      "checkpoint_revision": "1c5edc17a7acd8701df6fc341c0d179f1c62c982"
    },
    "kev": {
      "model": "Kev-0.8B",
      "checkpoint_revision": "54f4f8777356cd5bbbb6c6919c657f26e6f2f6d8"
    }
  },
  "laya_settings": {
    "source_commit": "42626c348753fbb17572a813127df2278a1ec527",
    "weights": "FP32",
    "native_cuda_autocast": "BF16",
    "max_length": 512,
    "head_max_length": 192,
    "native_probability_decimals": 4,
    "temperature": "saved per question type and option count"
  },
  "kev_settings": {
    "source_commit": "5e94a28818cfd3d0ec9b8bca046dc8db0d79a704",
    "dtype": "BF16",
    "adapter_merged": false,
    "attn": "sdpa",
    "max_state": 8192,
    "max_branch": 8192,
    "date_facts": false,
    "prefix_cache": false,
    "native_probability_decimals": 2,
    "unrounded_probabilities": "retained only as a secondary diagnostic",
    "temperature": 2.406050072164233
  },
  "budget": {
    "user_reported_account_cap_usd": 13,
    "start_only_below_metered_and_billed_usd": 7,
    "stop_at_metered_or_billed_usd": 8.5,
    "independent_client_deadline_seconds": 1500,
    "startup_seconds": 600,
    "call_seconds": 900,
    "inference_seconds": 720,
    "min_containers": 0,
    "max_containers_per_class": 1,
    "scaledown_seconds": 30,
    "gpu": "L40S",
    "sequential_models": true,
    "max_ordinary_attempt_compute_usd": 0.97
  },
  "code_sha256": {
    "modal_baselines.py": "e0203dfadf76204c53fe000741868bd19650106975c89eb3a5f678997999d388",
    "run_benchmark.py": "851ad20f968c9ee21d08b148104dfde53906a1096e43b56551371c3d4aa75f28",
    "guard_run.py": "b67a1c60e6b8e4be6d2c47e59a362041ae57ea939fc7340010339aac5fa7a8e3"
  }
}
