{
  "status": "frozen_before_inference",
  "date": "2026-09-21",
  "open_jev_commit": "ed45657bf726c3b77408942830e5578f99df904e",
  "jevbench_commit": "75e6224ed8103bbc3485ca74820a2eaf7ce8abe0",
  "tasks": 231,
  "dataset_sha256": "dc3995d8ae1e2fc8e81ce38431add509eb8bb39b85aadfd0c7c32079382dde51",
  "file_hashes": {
    "original": "5c2414edb3006b8bfcb70fda433f0f9ca015759433849f8d3104328a1f7c4180",
    "easy": "231df3c2c8e88a1a8c137ebe85de96ba70fabd330849098ac7b3c52c70b7172b",
    "hard": "89e9e6becb33ed88c1de7d42dcc87531b2fb64cfaef4e1986faf7c37b3f80ebb"
  },
  "settings": {
    "gpu": "L40S",
    "dtype": "bfloat16",
    "candidate_batch_size": 1,
    "max_length": 16384,
    "prefix_cache": false
  },
  "measurement": "Four example warmups, five interleaved rounds of four examples, then all 231 public tasks serially. Server-side synchronized in-process wall time; excludes Modal RPC/network and model load. No retries or answer reordering. Raw failures retained as incorrect.",
  "scoring": "Unmodified JevBench score_task: strict sum 0.001, renormalization band 0.02; lexical tie break; score argmax accuracy and separate expected-value MAE.",
  "budget": {
    "account_cap_usd": 13,
    "stop_starting_runs_at_metered_or_billed_usd": 7,
    "max_model_runs": 2,
    "startup_timeout_seconds": 900,
    "call_timeout_seconds": 900,
    "inference_soft_deadline_seconds": 720,
    "gpu_rate_usd_per_hour": 1.95,
    "cpu_max_cores": 4,
    "memory_max_gib": 48,
    "max_run_compute_estimate_usd": 1.3,
    "min_containers": 0,
    "scaledown_seconds": 30,
    "client_absolute_deadline_seconds": 1500,
    "live_billing_stop_usd": 8.5
  },
  "planned_requests_per_model": 255
}
