{
  "status": "frozen_before_inference",
  "date": "2026-09-21",
  "dataset": "JevBench public subset",
  "phase": "public",
  "questions": 231,
  "dataset_sha256": "dc3995d8ae1e2fc8e81ce38431add509eb8bb39b85aadfd0c7c32079382dde51",
  "request_sha256": "f8e05ce96f39020d1cf9943c8252c5d1390509f7d150576118a348a30f869a81",
  "models": {
    "laya-typed": {
      "name": "Laya typed-decisions",
      "app": "dougdoes-laya-typed",
      "class": "LayaTyped",
      "method": "run",
      "revision": "f9ab0b228f0fc0f14d873dbc99038f135c2da1b2",
      "sha256": "4fa56de72383a9d3efa9cfa78955733c81b9fc8067a587ca4beb82c78107a24e"
    }
  },
  "family_counts": {
    "policy": 12,
    "intent": 24,
    "ordinal": 12,
    "extraction": 24,
    "adequacy": 12,
    "routing": 12,
    "fact": 12,
    "tool_selection": 12,
    "long_policy": 19,
    "probability": 10,
    "temporal_numeric": 15,
    "ambiguous": 7,
    "multi_hop": 18,
    "tradeoff": 6,
    "adversarial": 6,
    "trap": 8,
    "judge_hard": 17,
    "routing_hard": 5
  },
  "type_counts": {
    "noul": 74,
    "choice": 139,
    "score": 18
  },
  "source_code_sha256": {
    "modal_typed.py": "c905f7cbecb219957726b6e6e938242f637bc6c895e9d1a1c8e65a92f0a5d295",
    "run_suite.py": "307fe57c4979ea8370b25547f71b8eec03f4b41dea6f6f7a6ac5b9243fa6bade",
    "guard_suite.py": "042d59dc9875389352c1b3709f60bc6e98b39567c70b5312fe45f054c63dca24"
  },
  "method": "One serial response per question per model, native API probabilities, native saved calibration, no retries/selection/tuning. No dedicated warmup; containers may be reused between phases. Fixed denominator includes failures. Exact option order identical across models. JevBench score_task unmodified: strict sum0.001, permitted normalization0.02, lexical argmax ties. Score accuracy is argmax; expected-value MAE separate. Synchronized GPU-process latency excludes initialization/RPC/network and Laya/Kev separate input audit.",
  "budget": {
    "user_reported_account_cap_usd": 13,
    "start_below_metered_and_billed_usd": 7,
    "live_stop_usd": 8.5,
    "client_deadline_seconds": 1500,
    "sequential": true
  },
  "private_policy": "No training, checkpoint selection, calibration fitting, or task revisions based on these holdout outcomes. Private prompts/options/keys/rationales/IDs/responses stay outside public artifacts. Only allowlisted aggregate metrics and protocol commitment may be published."
}
