{
  "reason": "Original website Choice options were lists; Open-Jev requires a map. All 24 initial example/warmup requests were rejected before model inference. Kept their errors; preserved the first valid 231-task 2B benchmark stream. 9B benchmark receives only the same 231 valid benchmark requests. Both models then receive a separate corrected 24-request example stream.",
  "mapping": "Choice lists become ordered {option_text: null} maps. compile_request emits the exact original option text once, with no invented label prefix. State, instructions, answer strings and option order are unchanged.",
  "benchmark_requests_unchanged": true,
  "quality_run_warmup": "No successful example warmups before either primary benchmark. Model initialization excluded; first-call kernel overhead included.",
  "examples_measurement": "One warmup per scenario, then five interleaved rounds; separate from primary benchmark.",
  "budget": "$13 account cap; same $7 start, $8.50 live stop and 25-minute guardian per invocation. Two additional bounded example calls, at most 900+900 seconds per model; total planned upper envelope still below $13 including prior observed costs.",
  "completed_scored_benchmark_streams_per_model": 1,
  "earlier_transport_attempt_remote_progress": "Unknown; no saved responses, app stop verified."
}
