{
  "edition": "Bid Bench 1.0",
  "grading_version": "semantic-v4",
  "run_date": "September 28, 2026 UTC",
  "generated_at": "2026-09-28T16:27:26.614990+00:00",
  "delivery_tasks": 30,
  "repeats": 3,
  "models": 18,
  "total_attempts": 12610,
  "scheduled_records": 13920,
  "cancelled_jobs": 1310,
  "reported_attempts": 9450,
  "control_attempts": 3160,
  "download_json": "/benchmarks/bid-bench-1.0-2026-09-28.json",
  "download_csv": "/benchmarks/bid-bench-1.0-2026-09-28.csv",
  "scope": {
    "delivery": "30 controlled tasks: six per category, three repeats",
    "forecasts": "12 historical Texas projects, 240 candidate/project pairs, 16 item-price targets; retrospective reconstruction",
    "judgment": "Four unprompted blocked tasks, separate from delivery score",
    "interval": "95% case-cluster percentile bootstrap, 4,000 draws stratified by category; repeats kept within task",
    "limits": [
      "Small authored sample; uncertainty describes these cases, not contractor population.",
      "Structured extracted sources and schedule-based takeoffs, not scanned plans.",
      "Apps are tested JavaScript domain functions, not full apps.",
      "Forecast descriptor availability at original prediction date not independently verified.",
      "Twelve delivery attempts were interrupted by the test host; affected configurations are flagged and these are not reasoning failures.",
      "Primary-agent and automated validation; no independent human review."
    ]
  },
  "complete": true,
  "publication_ready": true,
  "configurations": [
    {
      "id": "openai/gpt-6-astra@low",
      "model_id": "openai/gpt-6-astra",
      "model_label": "GPT-6 Astra",
      "effort": "low",
      "label": "GPT-6 Astra \u00b7 Low",
      "short_label": "GPT-6 Astra Low",
      "score": 100.0,
      "passed": 90,
      "attempts": 90,
      "cost": 11.842277777777776,
      "tokens": 11621.144444444444,
      "latency": 10.73800843799999,
      "ci": [
        100.0,
        100.0
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 10.65805,
      "categories": {
        "Documents": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Stress": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 3.8947036460002415
      },
      "predictions": {
        "brier": {
          "metric": 0.06719941922222222,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.5716322164394324,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 90,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 90,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-6-astra@medium",
      "model_id": "openai/gpt-6-astra",
      "model_label": "GPT-6 Astra",
      "effort": "medium",
      "label": "GPT-6 Astra \u00b7 Medium",
      "short_label": "GPT-6 Astra Medium",
      "score": 100.0,
      "passed": 90,
      "attempts": 90,
      "cost": 13.959811666666665,
      "tokens": 12847.888888888889,
      "latency": 12.788350708500715,
      "ci": [
        100.0,
        100.0
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 12.5638305,
      "categories": {
        "Documents": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Stress": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        }
      },
      "judgment": {
        "correct": 11,
        "attempts": 12,
        "correct_rate": 91.66666666666667,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 4.049042833999963
      },
      "predictions": {
        "brier": {
          "metric": 0.06728328408333333,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.5399095273638022,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 90,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 90,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-6-astra@high",
      "model_id": "openai/gpt-6-astra",
      "model_label": "GPT-6 Astra",
      "effort": "high",
      "label": "GPT-6 Astra \u00b7 High",
      "short_label": "GPT-6 Astra High",
      "score": 100.0,
      "passed": 90,
      "attempts": 90,
      "cost": 17.87373611111111,
      "tokens": 13920.977777777778,
      "latency": 18.920124417000217,
      "ci": [
        100.0,
        100.0
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 16.0863625,
      "categories": {
        "Documents": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Stress": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 5.019656312499894
      },
      "predictions": {
        "brier": {
          "metric": 0.06758012848611111,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.5387447252675678,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 90,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 90,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-6-astra@xhigh",
      "model_id": "openai/gpt-6-astra",
      "model_label": "GPT-6 Astra",
      "effort": "xhigh",
      "label": "GPT-6 Astra \u00b7 XHigh",
      "short_label": "GPT-6 Astra XHigh",
      "score": 98.88888888888889,
      "passed": 89,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 27.73193104100041,
      "ci": [
        96.67,
        100.0
      ],
      "cost_known_attempts": 89,
      "tokens_known_attempts": 89,
      "cost_known_subtotal": 18.950629,
      "categories": {
        "Documents": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Stress": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 6.402764979000203
      },
      "predictions": {
        "brier": {
          "metric": 0.0674106542361111,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.5531448214814492,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 89,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 89,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-6-astra@max",
      "model_id": "openai/gpt-6-astra",
      "model_label": "GPT-6 Astra",
      "effort": "max",
      "label": "GPT-6 Astra \u00b7 Max",
      "short_label": "GPT-6 Astra Max",
      "score": 94.44444444444444,
      "passed": 85,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 31.477933207999914,
      "ci": [
        87.78,
        100.0
      ],
      "cost_known_attempts": 85,
      "tokens_known_attempts": 85,
      "cost_known_subtotal": 20.060546,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Frontier": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Stress": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 7.341566145499703
      },
      "predictions": {
        "brier": {
          "metric": 0.06754189144444445,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.5635214721417133,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 85,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 85,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-6-sol@none",
      "model_id": "openai/gpt-6-sol",
      "model_label": "GPT-6 Sol",
      "effort": "none",
      "label": "GPT-6 Sol \u00b7 Off",
      "short_label": "GPT-6 Sol Off",
      "score": 83.33333333333333,
      "passed": 75,
      "attempts": 90,
      "cost": 3.0795877333333332,
      "tokens": 17675.166666666668,
      "latency": 7.889356167000253,
      "ci": [
        73.33,
        92.22
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 2.3096908,
      "categories": {
        "Documents": {
          "passed": 11,
          "attempts": 18,
          "rate": 61.111111111111114
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Knowledge": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Complex": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Expert": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Frontier": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Stress": {
          "passed": 7,
          "attempts": 15,
          "rate": 46.666666666666664
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 3,
        "solvable_attempts": 90,
        "false_stop_rate": 3.3333333333333335,
        "median_correct_stop_s": 2.7702606459999224
      },
      "predictions": {
        "brier": {
          "metric": 0.07068222222222222,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.627929352824778,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 75,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 75,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-6-sol@low",
      "model_id": "openai/gpt-6-sol",
      "model_label": "GPT-6 Sol",
      "effort": "low",
      "label": "GPT-6 Sol \u00b7 Low",
      "short_label": "GPT-6 Sol Low",
      "score": 94.44444444444444,
      "passed": 85,
      "attempts": 90,
      "cost": 2.327591411764706,
      "tokens": 12034.477777777778,
      "latency": 7.382528458000161,
      "ci": [
        87.78,
        100.0
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 1.9784527,
      "categories": {
        "Documents": {
          "passed": 14,
          "attempts": 18,
          "rate": 77.77777777777777
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Stress": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 3,
        "solvable_attempts": 90,
        "false_stop_rate": 3.3333333333333335,
        "median_correct_stop_s": 3.6567647920001765
      },
      "predictions": {
        "brier": {
          "metric": 0.07098764166666667,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.6055948636176703,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 85,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 85,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-6-sol@medium",
      "model_id": "openai/gpt-6-sol",
      "model_label": "GPT-6 Sol",
      "effort": "medium",
      "label": "GPT-6 Sol \u00b7 Medium",
      "short_label": "GPT-6 Sol Medium",
      "score": 94.44444444444444,
      "passed": 85,
      "attempts": 90,
      "cost": 2.9137298823529414,
      "tokens": 13382.555555555555,
      "latency": 10.411654500000004,
      "ci": [
        86.67,
        100.0
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 2.4766704,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Stress": {
          "passed": 10,
          "attempts": 15,
          "rate": 66.66666666666667
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 3,
        "solvable_attempts": 90,
        "false_stop_rate": 3.3333333333333335,
        "median_correct_stop_s": 4.662467749999836
      },
      "predictions": {
        "brier": {
          "metric": 0.06821722916666667,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.631205072494237,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 85,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 85,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-6-sol@high",
      "model_id": "openai/gpt-6-sol",
      "model_label": "GPT-6 Sol",
      "effort": "high",
      "label": "GPT-6 Sol \u00b7 High",
      "short_label": "GPT-6 Sol High",
      "score": 97.77777777777777,
      "passed": 88,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 11.721683999999776,
      "ci": [
        93.33,
        100.0
      ],
      "cost_known_attempts": 88,
      "tokens_known_attempts": 88,
      "cost_known_subtotal": 2.9285398000000002,
      "categories": {
        "Documents": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Stress": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 4.895087416999973
      },
      "predictions": {
        "brier": {
          "metric": 0.06731681666666667,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.6168871885831155,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 88,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 88,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-6-sol@xhigh",
      "model_id": "openai/gpt-6-sol",
      "model_label": "GPT-6 Sol",
      "effort": "xhigh",
      "label": "GPT-6 Sol \u00b7 XHigh",
      "short_label": "GPT-6 Sol XHigh",
      "score": 95.55555555555556,
      "passed": 86,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 13.467996500500012,
      "ci": [
        88.89,
        100.0
      ],
      "cost_known_attempts": 87,
      "tokens_known_attempts": 87,
      "cost_known_subtotal": 2.9832802,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Stress": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 5.234220208000043
      },
      "predictions": {
        "brier": {
          "metric": 0.06779464816666667,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.6001520720351002,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 86,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 86,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-6-sol@max",
      "model_id": "openai/gpt-6-sol",
      "model_label": "GPT-6 Sol",
      "effort": "max",
      "label": "GPT-6 Sol \u00b7 Max",
      "short_label": "GPT-6 Sol Max",
      "score": 95.55555555555556,
      "passed": 86,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 15.759853583498858,
      "ci": [
        88.89,
        100.0
      ],
      "cost_known_attempts": 86,
      "tokens_known_attempts": 86,
      "cost_known_subtotal": 3.9076325,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Stress": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 7.002121104000136
      },
      "predictions": {
        "brier": {
          "metric": 0.06746843273611111,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.6134672153994195,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 86,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 86,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-6-luna@none",
      "model_id": "openai/gpt-6-luna",
      "model_label": "GPT-6 Luna",
      "effort": "none",
      "label": "GPT-6 Luna \u00b7 Off",
      "short_label": "GPT-6 Luna Off",
      "score": 56.666666666666664,
      "passed": 51,
      "attempts": 90,
      "cost": 0.25696016666666666,
      "tokens": 20710.577777777777,
      "latency": 7.9663701249999,
      "ci": [
        42.22,
        71.11
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 0.131049685,
      "categories": {
        "Documents": {
          "passed": 8,
          "attempts": 18,
          "rate": 44.44444444444444
        },
        "Takeoffs": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Analysis": {
          "passed": 6,
          "attempts": 18,
          "rate": 33.333333333333336
        },
        "Knowledge": {
          "passed": 7,
          "attempts": 18,
          "rate": 38.888888888888886
        },
        "Apps": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Multi-step": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Complex": {
          "passed": 9,
          "attempts": 15,
          "rate": 60.0
        },
        "Expert": {
          "passed": 9,
          "attempts": 15,
          "rate": 60.0
        },
        "Frontier": {
          "passed": 6,
          "attempts": 15,
          "rate": 40.0
        },
        "Stress": {
          "passed": 1,
          "attempts": 15,
          "rate": 6.666666666666667
        }
      },
      "judgment": {
        "correct": 9,
        "attempts": 12,
        "correct_rate": 75.0,
        "false_stops": 12,
        "solvable_attempts": 90,
        "false_stop_rate": 13.333333333333334,
        "median_correct_stop_s": 1.7103761670002713
      },
      "predictions": {
        "brier": {
          "metric": 0.1089188,
          "baseline_matched": 0.06915372239178526,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 8,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.4471104988691634,
          "baseline_matched": 0.3771298981519376,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 10,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 51,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 51,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-6-luna@low",
      "model_id": "openai/gpt-6-luna",
      "model_label": "GPT-6 Luna",
      "effort": "low",
      "label": "GPT-6 Luna \u00b7 Low",
      "short_label": "GPT-6 Luna Low",
      "score": 70.0,
      "passed": 63,
      "attempts": 90,
      "cost": 0.1687146507936508,
      "tokens": 8449.888888888889,
      "latency": 6.8721844159998,
      "ci": [
        56.67,
        83.33
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 0.10629023,
      "categories": {
        "Documents": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Takeoffs": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        },
        "Analysis": {
          "passed": 9,
          "attempts": 18,
          "rate": 50.0
        },
        "Knowledge": {
          "passed": 10,
          "attempts": 18,
          "rate": 55.55555555555556
        },
        "Apps": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Complex": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        },
        "Expert": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        },
        "Frontier": {
          "passed": 9,
          "attempts": 15,
          "rate": 60.0
        },
        "Stress": {
          "passed": 5,
          "attempts": 15,
          "rate": 33.333333333333336
        }
      },
      "judgment": {
        "correct": 11,
        "attempts": 12,
        "correct_rate": 91.66666666666667,
        "false_stops": 10,
        "solvable_attempts": 90,
        "false_stop_rate": 11.11111111111111,
        "median_correct_stop_s": 3.589884166999953
      },
      "predictions": {
        "brier": {
          "metric": 0.09586666666666667,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.4176012456588276,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 63,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 63,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-6-luna@medium",
      "model_id": "openai/gpt-6-luna",
      "model_label": "GPT-6 Luna",
      "effort": "medium",
      "label": "GPT-6 Luna \u00b7 Medium",
      "short_label": "GPT-6 Luna Medium",
      "score": 78.88888888888889,
      "passed": 71,
      "attempts": 90,
      "cost": 0.217867838028169,
      "tokens": 15802.122222222222,
      "latency": 9.610592874999973,
      "ci": [
        68.89,
        88.89
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 0.154686165,
      "categories": {
        "Documents": {
          "passed": 13,
          "attempts": 18,
          "rate": 72.22222222222223
        },
        "Takeoffs": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Analysis": {
          "passed": 9,
          "attempts": 18,
          "rate": 50.0
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Multi-step": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Complex": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Expert": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Frontier": {
          "passed": 9,
          "attempts": 15,
          "rate": 60.0
        },
        "Stress": {
          "passed": 7,
          "attempts": 15,
          "rate": 46.666666666666664
        }
      },
      "judgment": {
        "correct": 10,
        "attempts": 12,
        "correct_rate": 83.33333333333333,
        "false_stops": 3,
        "solvable_attempts": 90,
        "false_stop_rate": 3.3333333333333335,
        "median_correct_stop_s": 3.465215187499882
      },
      "predictions": {
        "brier": {
          "metric": 0.06885169861111111,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.44579599858238617,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 71,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 71,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-6-luna@high",
      "model_id": "openai/gpt-6-luna",
      "model_label": "GPT-6 Luna",
      "effort": "high",
      "label": "GPT-6 Luna \u00b7 High",
      "short_label": "GPT-6 Luna High",
      "score": 78.88888888888889,
      "passed": 71,
      "attempts": 90,
      "cost": 0.25143735211267604,
      "tokens": 15844.211111111112,
      "latency": 9.951377582999879,
      "ci": [
        68.89,
        88.89
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 0.17852052000000002,
      "categories": {
        "Documents": {
          "passed": 11,
          "attempts": 18,
          "rate": 61.111111111111114
        },
        "Takeoffs": {
          "passed": 14,
          "attempts": 18,
          "rate": 77.77777777777777
        },
        "Analysis": {
          "passed": 11,
          "attempts": 18,
          "rate": 61.111111111111114
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Multi-step": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Complex": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Expert": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Frontier": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Stress": {
          "passed": 8,
          "attempts": 15,
          "rate": 53.333333333333336
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 3,
        "solvable_attempts": 90,
        "false_stop_rate": 3.3333333333333335,
        "median_correct_stop_s": 3.7464250415000135
      },
      "predictions": {
        "brier": {
          "metric": 0.06804867916666667,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.5875202703677846,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 70,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 70,
      "semantic_changed_delivery": 1,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-6-luna@xhigh",
      "model_id": "openai/gpt-6-luna",
      "model_label": "GPT-6 Luna",
      "effort": "xhigh",
      "label": "GPT-6 Luna \u00b7 XHigh",
      "short_label": "GPT-6 Luna XHigh",
      "score": 88.88888888888889,
      "passed": 80,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 11.530144562500006,
      "ci": [
        80.0,
        96.67
      ],
      "cost_known_attempts": 88,
      "tokens_known_attempts": 88,
      "cost_known_subtotal": 0.20797535,
      "categories": {
        "Documents": {
          "passed": 14,
          "attempts": 18,
          "rate": 77.77777777777777
        },
        "Takeoffs": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        },
        "Analysis": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Frontier": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Stress": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 1,
        "solvable_attempts": 90,
        "false_stop_rate": 1.1111111111111112,
        "median_correct_stop_s": 4.5776513334997
      },
      "predictions": {
        "brier": {
          "metric": 0.06814458626388889,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.5285964073207607,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 80,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 80,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-6-luna@max",
      "model_id": "openai/gpt-6-luna",
      "model_label": "GPT-6 Luna",
      "effort": "max",
      "label": "GPT-6 Luna \u00b7 Max",
      "short_label": "GPT-6 Luna Max",
      "score": 92.22222222222223,
      "passed": 83,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 18.15507424999995,
      "ci": [
        83.33,
        98.89
      ],
      "cost_known_attempts": 87,
      "tokens_known_attempts": 87,
      "cost_known_subtotal": 0.306733895,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Expert": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Frontier": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Stress": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 6.078004895500141
      },
      "predictions": {
        "brier": {
          "metric": 0.06870528068055556,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.5425132827351865,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 83,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 83,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-5.6-terra@none",
      "model_id": "openai/gpt-5.6-terra",
      "model_label": "GPT-5.6 Terra",
      "effort": "none",
      "label": "GPT-5.6 Terra \u00b7 Off",
      "short_label": "GPT-5.6 Terra Off",
      "score": 65.55555555555556,
      "passed": 59,
      "attempts": 90,
      "cost": 5.007509661016949,
      "tokens": 13513.977777777778,
      "latency": 4.674105959000066,
      "ci": [
        53.33,
        78.89
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 2.9544307,
      "categories": {
        "Documents": {
          "passed": 8,
          "attempts": 18,
          "rate": 44.44444444444444
        },
        "Takeoffs": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Analysis": {
          "passed": 7,
          "attempts": 18,
          "rate": 38.888888888888886
        },
        "Knowledge": {
          "passed": 10,
          "attempts": 18,
          "rate": 55.55555555555556
        },
        "Apps": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 8,
          "attempts": 15,
          "rate": 53.333333333333336
        },
        "Expert": {
          "passed": 10,
          "attempts": 15,
          "rate": 66.66666666666667
        },
        "Frontier": {
          "passed": 6,
          "attempts": 15,
          "rate": 40.0
        },
        "Stress": {
          "passed": 6,
          "attempts": 15,
          "rate": 40.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 9,
        "solvable_attempts": 90,
        "false_stop_rate": 10.0,
        "median_correct_stop_s": 2.044341083000065
      },
      "predictions": {
        "brier": {
          "metric": 0.09526208333333333,
          "baseline_matched": 0.07859125554029071,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 8,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.6672198057010744,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 59,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 59,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-5.6-terra@low",
      "model_id": "openai/gpt-5.6-terra",
      "model_label": "GPT-5.6 Terra",
      "effort": "low",
      "label": "GPT-5.6 Terra \u00b7 Low",
      "short_label": "GPT-5.6 Terra Low",
      "score": 85.55555555555556,
      "passed": 77,
      "attempts": 90,
      "cost": 3.772196493506493,
      "tokens": 15669.111111111111,
      "latency": 8.616859749999945,
      "ci": [
        76.67,
        93.33
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 2.9045913,
      "categories": {
        "Documents": {
          "passed": 11,
          "attempts": 18,
          "rate": 61.111111111111114
        },
        "Takeoffs": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Analysis": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Knowledge": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Apps": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        },
        "Stress": {
          "passed": 9,
          "attempts": 15,
          "rate": 60.0
        }
      },
      "judgment": {
        "correct": 9,
        "attempts": 12,
        "correct_rate": 75.0,
        "false_stops": 3,
        "solvable_attempts": 90,
        "false_stop_rate": 3.3333333333333335,
        "median_correct_stop_s": 2.938163832999766
      },
      "predictions": {
        "brier": {
          "metric": 0.06842101569444445,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.4422145684544501,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 74,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 74,
      "semantic_changed_delivery": 3,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-5.6-terra@medium",
      "model_id": "openai/gpt-5.6-terra",
      "model_label": "GPT-5.6 Terra",
      "effort": "medium",
      "label": "GPT-5.6 Terra \u00b7 Medium",
      "short_label": "GPT-5.6 Terra Medium",
      "score": 86.66666666666667,
      "passed": 78,
      "attempts": 90,
      "cost": 4.340710769230769,
      "tokens": 17737.433333333334,
      "latency": 9.376397604000289,
      "ci": [
        77.78,
        94.44
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 3.3857544,
      "categories": {
        "Documents": {
          "passed": 13,
          "attempts": 18,
          "rate": 72.22222222222223
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 14,
          "attempts": 18,
          "rate": 77.77777777777777
        },
        "Knowledge": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Apps": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Complex": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Stress": {
          "passed": 8,
          "attempts": 15,
          "rate": 53.333333333333336
        }
      },
      "judgment": {
        "correct": 9,
        "attempts": 12,
        "correct_rate": 75.0,
        "false_stops": 1,
        "solvable_attempts": 90,
        "false_stop_rate": 1.1111111111111112,
        "median_correct_stop_s": 3.558902499999851
      },
      "predictions": {
        "brier": {
          "metric": 0.06812799253030304,
          "baseline_matched": 0.06513735513100806,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 11,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.4121465637860785,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 76,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 76,
      "semantic_changed_delivery": 2,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-5.6-terra@high",
      "model_id": "openai/gpt-5.6-terra",
      "model_label": "GPT-5.6 Terra",
      "effort": "high",
      "label": "GPT-5.6 Terra \u00b7 High",
      "short_label": "GPT-5.6 Terra High",
      "score": 94.44444444444444,
      "passed": 85,
      "attempts": 90,
      "cost": 4.932232588235293,
      "tokens": 19705.144444444446,
      "latency": 10.906350084000268,
      "ci": [
        87.78,
        100.0
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 4.1923977,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Knowledge": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Stress": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        }
      },
      "judgment": {
        "correct": 11,
        "attempts": 12,
        "correct_rate": 91.66666666666667,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 3.31175841699983
      },
      "predictions": {
        "brier": {
          "metric": 0.06913299877777777,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.4276202532618688,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 85,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 85,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-5.6-terra@xhigh",
      "model_id": "openai/gpt-5.6-terra",
      "model_label": "GPT-5.6 Terra",
      "effort": "xhigh",
      "label": "GPT-5.6 Terra \u00b7 XHigh",
      "short_label": "GPT-5.6 Terra XHigh",
      "score": 87.77777777777777,
      "passed": 79,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 11.536145499999636,
      "ci": [
        78.89,
        94.44
      ],
      "cost_known_attempts": 88,
      "tokens_known_attempts": 88,
      "cost_known_subtotal": 4.1921852,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Analysis": {
          "passed": 14,
          "attempts": 18,
          "rate": 77.77777777777777
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Frontier": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Stress": {
          "passed": 9,
          "attempts": 15,
          "rate": 60.0
        }
      },
      "judgment": {
        "correct": 10,
        "attempts": 12,
        "correct_rate": 83.33333333333333,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 3.4898537500002424
      },
      "predictions": {
        "brier": {
          "metric": 0.06500877325757576,
          "baseline_matched": 0.06513735513100806,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 11,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.5423540791917535,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 78,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 78,
      "semantic_changed_delivery": 1,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-5.6-terra@max",
      "model_id": "openai/gpt-5.6-terra",
      "model_label": "GPT-5.6 Terra",
      "effort": "max",
      "label": "GPT-5.6 Terra \u00b7 Max",
      "short_label": "GPT-5.6 Terra Max",
      "score": 92.22222222222223,
      "passed": 83,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 19.287076500000026,
      "ci": [
        84.44,
        97.78
      ],
      "cost_known_attempts": 84,
      "tokens_known_attempts": 84,
      "cost_known_subtotal": 6.8606128,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 14,
          "attempts": 18,
          "rate": 77.77777777777777
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Stress": {
          "passed": 10,
          "attempts": 15,
          "rate": 66.66666666666667
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 4.9153228125002935
      },
      "predictions": {
        "brier": {
          "metric": 0.07193003719755209,
          "baseline_matched": 0.07252651287538876,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 8,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.5231404162600766,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 83,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 83,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-5.6-luna@none",
      "model_id": "openai/gpt-5.6-luna",
      "model_label": "GPT-5.6 Luna",
      "effort": "none",
      "label": "GPT-5.6 Luna \u00b7 Off",
      "short_label": "GPT-5.6 Luna Off",
      "score": 56.666666666666664,
      "passed": 51,
      "attempts": 90,
      "cost": 0.5673087647058823,
      "tokens": 13522.822222222223,
      "latency": 4.602526041001082,
      "ci": [
        43.33,
        71.11
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 0.28932747,
      "categories": {
        "Documents": {
          "passed": 14,
          "attempts": 18,
          "rate": 77.77777777777777
        },
        "Takeoffs": {
          "passed": 9,
          "attempts": 18,
          "rate": 50.0
        },
        "Analysis": {
          "passed": 5,
          "attempts": 18,
          "rate": 27.77777777777778
        },
        "Knowledge": {
          "passed": 10,
          "attempts": 18,
          "rate": 55.55555555555556
        },
        "Apps": {
          "passed": 13,
          "attempts": 18,
          "rate": 72.22222222222223
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Multi-step": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Complex": {
          "passed": 8,
          "attempts": 15,
          "rate": 53.333333333333336
        },
        "Expert": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        },
        "Frontier": {
          "passed": 7,
          "attempts": 15,
          "rate": 46.666666666666664
        },
        "Stress": {
          "passed": 0,
          "attempts": 15,
          "rate": 0.0
        }
      },
      "judgment": {
        "correct": 9,
        "attempts": 12,
        "correct_rate": 75.0,
        "false_stops": 13,
        "solvable_attempts": 90,
        "false_stop_rate": 14.444444444444445,
        "median_correct_stop_s": 1.9414729579999113
      },
      "predictions": {
        "brier": {
          "metric": 0.09055862799742084,
          "baseline_matched": 0.061389102494270285,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 8,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.40330963086655397,
          "baseline_matched": 0.37683259193474544,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 8,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 51,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 51,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-5.6-luna@low",
      "model_id": "openai/gpt-5.6-luna",
      "model_label": "GPT-5.6 Luna",
      "effort": "low",
      "label": "GPT-5.6 Luna \u00b7 Low",
      "short_label": "GPT-5.6 Luna Low",
      "score": 65.55555555555556,
      "passed": 59,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 6.3965325830001385,
      "ci": [
        54.44,
        75.56
      ],
      "cost_known_attempts": 89,
      "tokens_known_attempts": 89,
      "cost_known_subtotal": 0.25709461,
      "categories": {
        "Documents": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Takeoffs": {
          "passed": 5,
          "attempts": 18,
          "rate": 27.77777777777778
        },
        "Analysis": {
          "passed": 11,
          "attempts": 18,
          "rate": 61.111111111111114
        },
        "Knowledge": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Apps": {
          "passed": 14,
          "attempts": 18,
          "rate": 77.77777777777777
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Multi-step": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Complex": {
          "passed": 10,
          "attempts": 15,
          "rate": 66.66666666666667
        },
        "Expert": {
          "passed": 10,
          "attempts": 15,
          "rate": 66.66666666666667
        },
        "Frontier": {
          "passed": 8,
          "attempts": 15,
          "rate": 53.333333333333336
        },
        "Stress": {
          "passed": 7,
          "attempts": 15,
          "rate": 46.666666666666664
        }
      },
      "judgment": {
        "correct": 10,
        "attempts": 12,
        "correct_rate": 83.33333333333333,
        "false_stops": 4,
        "solvable_attempts": 90,
        "false_stop_rate": 4.444444444444445,
        "median_correct_stop_s": 3.488273291500169
      },
      "predictions": {
        "brier": {
          "metric": 0.11708950295,
          "baseline_matched": 0.0691711721070809,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 10,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.35679192192933584,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 58,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 58,
      "semantic_changed_delivery": 1,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-5.6-luna@medium",
      "model_id": "openai/gpt-5.6-luna",
      "model_label": "GPT-5.6 Luna",
      "effort": "medium",
      "label": "GPT-5.6 Luna \u00b7 Medium",
      "short_label": "GPT-5.6 Luna Medium",
      "score": 73.33333333333333,
      "passed": 66,
      "attempts": 90,
      "cost": 0.4634262727272727,
      "tokens": 12353.755555555555,
      "latency": 13.893788874999998,
      "ci": [
        61.11,
        85.56
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 0.30586134,
      "categories": {
        "Documents": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Takeoffs": {
          "passed": 9,
          "attempts": 18,
          "rate": 50.0
        },
        "Analysis": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Knowledge": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        },
        "Apps": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Multi-step": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Complex": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        },
        "Expert": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Frontier": {
          "passed": 8,
          "attempts": 15,
          "rate": 53.333333333333336
        },
        "Stress": {
          "passed": 10,
          "attempts": 15,
          "rate": 66.66666666666667
        }
      },
      "judgment": {
        "correct": 9,
        "attempts": 12,
        "correct_rate": 75.0,
        "false_stops": 3,
        "solvable_attempts": 90,
        "false_stop_rate": 3.3333333333333335,
        "median_correct_stop_s": 4.1791458329998425
      },
      "predictions": {
        "brier": {
          "metric": 0.07235710118055555,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.35679192192933584,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 66,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 66,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-5.6-luna@high",
      "model_id": "openai/gpt-5.6-luna",
      "model_label": "GPT-5.6 Luna",
      "effort": "high",
      "label": "GPT-5.6 Luna \u00b7 High",
      "short_label": "GPT-5.6 Luna High",
      "score": 78.88888888888889,
      "passed": 71,
      "attempts": 90,
      "cost": 0.8373292816901409,
      "tokens": 27833.766666666666,
      "latency": 16.04657466699928,
      "ci": [
        71.11,
        86.67
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 0.59450379,
      "categories": {
        "Documents": {
          "passed": 11,
          "attempts": 18,
          "rate": 61.111111111111114
        },
        "Takeoffs": {
          "passed": 8,
          "attempts": 18,
          "rate": 44.44444444444444
        },
        "Analysis": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Multi-step": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Complex": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Expert": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        },
        "Frontier": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Stress": {
          "passed": 10,
          "attempts": 15,
          "rate": 66.66666666666667
        }
      },
      "judgment": {
        "correct": 10,
        "attempts": 12,
        "correct_rate": 83.33333333333333,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 3.5176495834997623
      },
      "predictions": {
        "brier": {
          "metric": 0.070265128427295,
          "baseline_matched": 0.06962447270244437,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 10,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.3566715849858942,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 68,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 68,
      "semantic_changed_delivery": 3,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-5.6-luna@xhigh",
      "model_id": "openai/gpt-5.6-luna",
      "model_label": "GPT-5.6 Luna",
      "effort": "xhigh",
      "label": "GPT-5.6 Luna \u00b7 XHigh",
      "short_label": "GPT-5.6 Luna XHigh",
      "score": 81.11111111111111,
      "passed": 73,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 13.629914792000026,
      "ci": [
        71.11,
        90.0
      ],
      "cost_known_attempts": 89,
      "tokens_known_attempts": 89,
      "cost_known_subtotal": 0.63698831,
      "categories": {
        "Documents": {
          "passed": 13,
          "attempts": 18,
          "rate": 72.22222222222223
        },
        "Takeoffs": {
          "passed": 9,
          "attempts": 18,
          "rate": 50.0
        },
        "Analysis": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Multi-step": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Complex": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Expert": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Frontier": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Stress": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        }
      },
      "judgment": {
        "correct": 10,
        "attempts": 12,
        "correct_rate": 83.33333333333333,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 3.868886832999764
      },
      "predictions": {
        "brier": {
          "metric": 0.06548940015425667,
          "baseline_matched": 0.0659438330808992,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 5,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.3566715849858942,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 70,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 70,
      "semantic_changed_delivery": 3,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-5.6-luna@max",
      "model_id": "openai/gpt-5.6-luna",
      "model_label": "GPT-5.6 Luna",
      "effort": "max",
      "label": "GPT-5.6 Luna \u00b7 Max",
      "short_label": "GPT-5.6 Luna Max",
      "score": 78.88888888888889,
      "passed": 71,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 27.322540416999953,
      "ci": [
        67.78,
        88.89
      ],
      "cost_known_attempts": 87,
      "tokens_known_attempts": 87,
      "cost_known_subtotal": 0.75089451,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 9,
          "attempts": 18,
          "rate": 50.0
        },
        "Analysis": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 13,
          "attempts": 18,
          "rate": 72.22222222222223
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Multi-step": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        },
        "Complex": {
          "passed": 10,
          "attempts": 15,
          "rate": 66.66666666666667
        },
        "Expert": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Frontier": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Stress": {
          "passed": 10,
          "attempts": 15,
          "rate": 66.66666666666667
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 5.097012853500084
      },
      "predictions": {
        "brier": {
          "metric": 0.07407973772350833,
          "baseline_matched": 0.07285720452495058,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 8,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.35679192192933584,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 71,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 71,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-5.6-sol@none",
      "model_id": "openai/gpt-5.6-sol",
      "model_label": "GPT-5.6 Sol",
      "effort": "none",
      "label": "GPT-5.6 Sol \u00b7 Off",
      "short_label": "GPT-5.6 Sol Off",
      "score": 84.44444444444444,
      "passed": 76,
      "attempts": 90,
      "cost": 7.841352894736842,
      "tokens": 19565.633333333335,
      "latency": 9.090221458000014,
      "ci": [
        73.33,
        94.44
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 5.9594282,
      "categories": {
        "Documents": {
          "passed": 13,
          "attempts": 18,
          "rate": 72.22222222222223
        },
        "Takeoffs": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        },
        "Analysis": {
          "passed": 13,
          "attempts": 18,
          "rate": 72.22222222222223
        },
        "Knowledge": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        },
        "Stress": {
          "passed": 5,
          "attempts": 15,
          "rate": 33.333333333333336
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 5,
        "solvable_attempts": 90,
        "false_stop_rate": 5.555555555555555,
        "median_correct_stop_s": 2.031696708000032
      },
      "predictions": {
        "brier": {
          "metric": 0.11856569454166667,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.43034103471524443,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 76,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 76,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-5.6-sol@low",
      "model_id": "openai/gpt-5.6-sol",
      "model_label": "GPT-5.6 Sol",
      "effort": "low",
      "label": "GPT-5.6 Sol \u00b7 Low",
      "short_label": "GPT-5.6 Sol Low",
      "score": 95.55555555555556,
      "passed": 86,
      "attempts": 90,
      "cost": 6.241331627906978,
      "tokens": 16797.31111111111,
      "latency": 13.098202687500452,
      "ci": [
        88.89,
        100.0
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 5.3675452,
      "categories": {
        "Documents": {
          "passed": 14,
          "attempts": 18,
          "rate": 77.77777777777777
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Stress": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 3,
        "solvable_attempts": 90,
        "false_stop_rate": 3.3333333333333335,
        "median_correct_stop_s": 3.945015375499846
      },
      "predictions": {
        "brier": {
          "metric": 0.06757576527777778,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.6066352858767079,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 86,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 86,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-5.6-sol@medium",
      "model_id": "openai/gpt-5.6-sol",
      "model_label": "GPT-5.6 Sol",
      "effort": "medium",
      "label": "GPT-5.6 Sol \u00b7 Medium",
      "short_label": "GPT-5.6 Sol Medium",
      "score": 96.66666666666667,
      "passed": 87,
      "attempts": 90,
      "cost": 8.701733103448275,
      "tokens": 21280.3,
      "latency": 14.350415917000268,
      "ci": [
        90.0,
        100.0
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 7.5705078,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Stress": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        }
      },
      "judgment": {
        "correct": 10,
        "attempts": 12,
        "correct_rate": 83.33333333333333,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 4.164828916499857
      },
      "predictions": {
        "brier": {
          "metric": 0.06779102477777778,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.5781923404890522,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 87,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 87,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-5.6-sol@high",
      "model_id": "openai/gpt-5.6-sol",
      "model_label": "GPT-5.6 Sol",
      "effort": "high",
      "label": "GPT-5.6 Sol \u00b7 High",
      "short_label": "GPT-5.6 Sol High",
      "score": 100.0,
      "passed": 90,
      "attempts": 90,
      "cost": 9.831807333333332,
      "tokens": 21905.98888888889,
      "latency": 17.646755333499982,
      "ci": [
        100.0,
        100.0
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 8.8486266,
      "categories": {
        "Documents": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Stress": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        }
      },
      "judgment": {
        "correct": 11,
        "attempts": 12,
        "correct_rate": 91.66666666666667,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 5.001593875000253
      },
      "predictions": {
        "brier": {
          "metric": 0.06809089473611112,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.588438939128508,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 90,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 90,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-5.6-sol@xhigh",
      "model_id": "openai/gpt-5.6-sol",
      "model_label": "GPT-5.6 Sol",
      "effort": "xhigh",
      "label": "GPT-5.6 Sol \u00b7 XHigh",
      "short_label": "GPT-5.6 Sol XHigh",
      "score": 100.0,
      "passed": 90,
      "attempts": 90,
      "cost": 11.66358311111111,
      "tokens": 26419.244444444445,
      "latency": 16.776120916499963,
      "ci": [
        100.0,
        100.0
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 10.4972248,
      "categories": {
        "Documents": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Stress": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        }
      },
      "judgment": {
        "correct": 11,
        "attempts": 12,
        "correct_rate": 91.66666666666667,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 3.817218708000146
      },
      "predictions": {
        "brier": {
          "metric": 0.06800736801685417,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.5781923404890522,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 90,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 90,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "openai/gpt-5.6-sol@max",
      "model_id": "openai/gpt-5.6-sol",
      "model_label": "GPT-5.6 Sol",
      "effort": "max",
      "label": "GPT-5.6 Sol \u00b7 Max",
      "short_label": "GPT-5.6 Sol Max",
      "score": 96.66666666666667,
      "passed": 87,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 17.094765499999745,
      "ci": [
        93.33,
        100.0
      ],
      "cost_known_attempts": 89,
      "tokens_known_attempts": 89,
      "cost_known_subtotal": 11.8921854,
      "categories": {
        "Documents": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Complex": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Stress": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 4.7536205420001645
      },
      "predictions": {
        "brier": {
          "metric": 0.06449371586666666,
          "baseline_matched": 0.06413063069944532,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 5,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.588438939128508,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 87,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 87,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "anthropic/claude-sonnet-5@none",
      "model_id": "anthropic/claude-sonnet-5",
      "model_label": "Claude Sonnet 5",
      "effort": "none",
      "label": "Claude Sonnet 5 \u00b7 Off",
      "short_label": "Claude Sonnet 5 Off",
      "score": 81.11111111111111,
      "passed": 73,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 39.13006308300048,
      "ci": [
        68.89,
        92.22
      ],
      "cost_known_attempts": 80,
      "tokens_known_attempts": 80,
      "cost_known_subtotal": 9.63843,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Knowledge": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Apps": {
          "passed": 11,
          "attempts": 18,
          "rate": 61.111111111111114
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Frontier": {
          "passed": 9,
          "attempts": 15,
          "rate": 60.0
        },
        "Stress": {
          "passed": 6,
          "attempts": 15,
          "rate": 40.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 7.933556166499621
      },
      "predictions": {
        "brier": {
          "metric": 0.14568454166666667,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.6426462584752479,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 73,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 73,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "anthropic/claude-sonnet-5@low",
      "model_id": "anthropic/claude-sonnet-5",
      "model_label": "Claude Sonnet 5",
      "effort": "low",
      "label": "Claude Sonnet 5 \u00b7 Low",
      "short_label": "Claude Sonnet 5 Low",
      "score": 85.55555555555556,
      "passed": 77,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 49.27621608299948,
      "ci": [
        74.44,
        95.56
      ],
      "cost_known_attempts": 82,
      "tokens_known_attempts": 82,
      "cost_known_subtotal": 11.335496,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Knowledge": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        },
        "Apps": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 10,
          "attempts": 15,
          "rate": 66.66666666666667
        },
        "Stress": {
          "passed": 9,
          "attempts": 15,
          "rate": 60.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 15.149287625000348
      },
      "predictions": {
        "brier": {
          "metric": 0.20722572083333332,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.5678543898398523,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 77,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 77,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "anthropic/claude-sonnet-5@medium",
      "model_id": "anthropic/claude-sonnet-5",
      "model_label": "Claude Sonnet 5",
      "effort": "medium",
      "label": "Claude Sonnet 5 \u00b7 Medium",
      "short_label": "Claude Sonnet 5 Medium",
      "score": 81.11111111111111,
      "passed": 73,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 51.74823729199916,
      "ci": [
        68.89,
        92.22
      ],
      "cost_known_attempts": 81,
      "tokens_known_attempts": 81,
      "cost_known_subtotal": 9.941956,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Analysis": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Knowledge": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        },
        "Apps": {
          "passed": 13,
          "attempts": 18,
          "rate": 72.22222222222223
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Frontier": {
          "passed": 7,
          "attempts": 15,
          "rate": 46.666666666666664
        },
        "Stress": {
          "passed": 8,
          "attempts": 15,
          "rate": 53.333333333333336
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 14.847921645499882
      },
      "predictions": {
        "brier": {
          "metric": 0.11225826111111112,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.6489667403829927,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 73,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 73,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "anthropic/claude-sonnet-5@high",
      "model_id": "anthropic/claude-sonnet-5",
      "model_label": "Claude Sonnet 5",
      "effort": "high",
      "label": "Claude Sonnet 5 \u00b7 High",
      "short_label": "Claude Sonnet 5 High",
      "score": 78.88888888888889,
      "passed": 71,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 41.34323358299956,
      "ci": [
        66.67,
        90.0
      ],
      "cost_known_attempts": 81,
      "tokens_known_attempts": 81,
      "cost_known_subtotal": 9.689196,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Analysis": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Knowledge": {
          "passed": 14,
          "attempts": 18,
          "rate": 77.77777777777777
        },
        "Apps": {
          "passed": 13,
          "attempts": 18,
          "rate": 72.22222222222223
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Expert": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Frontier": {
          "passed": 8,
          "attempts": 15,
          "rate": 53.333333333333336
        },
        "Stress": {
          "passed": 7,
          "attempts": 15,
          "rate": 46.666666666666664
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 14.656794812499895
      },
      "predictions": {
        "brier": {
          "metric": 0.17964797777777777,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.6423620522227104,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 71,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 71,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "anthropic/claude-sonnet-5@xhigh",
      "model_id": "anthropic/claude-sonnet-5",
      "model_label": "Claude Sonnet 5",
      "effort": "xhigh",
      "label": "Claude Sonnet 5 \u00b7 XHigh",
      "short_label": "Claude Sonnet 5 XHigh",
      "score": 76.66666666666667,
      "passed": 69,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 42.774537291000016,
      "ci": [
        64.44,
        87.78
      ],
      "cost_known_attempts": 80,
      "tokens_known_attempts": 80,
      "cost_known_subtotal": 9.001286,
      "categories": {
        "Documents": {
          "passed": 14,
          "attempts": 18,
          "rate": 77.77777777777777
        },
        "Takeoffs": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Analysis": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Knowledge": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        },
        "Apps": {
          "passed": 10,
          "attempts": 18,
          "rate": 55.55555555555556
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Complex": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Expert": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Frontier": {
          "passed": 8,
          "attempts": 15,
          "rate": 53.333333333333336
        },
        "Stress": {
          "passed": 7,
          "attempts": 15,
          "rate": 46.666666666666664
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 15.25878341699997
      },
      "predictions": {
        "brier": {
          "metric": 0.17246157083333333,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.6442052895176283,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 69,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 69,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 1
    },
    {
      "id": "anthropic/claude-fable-5.1@none",
      "model_id": "anthropic/claude-fable-5.1",
      "model_label": "Claude Fable 5.1",
      "effort": "none",
      "label": "Claude Fable 5.1 \u00b7 Off",
      "short_label": "Claude Fable 5.1 Off",
      "score": 92.22222222222223,
      "passed": 83,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 33.467916042000056,
      "ci": [
        83.31,
        100.0
      ],
      "cost_known_attempts": 83,
      "tokens_known_attempts": 83,
      "cost_known_subtotal": 26.6832,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 14,
          "attempts": 18,
          "rate": 77.77777777777777
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Stress": {
          "passed": 9,
          "attempts": 15,
          "rate": 60.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 9.153811041499836
      },
      "predictions": {
        "brier": {
          "metric": 0.0683206125,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.47593169449337547,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 83,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 83,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "anthropic/claude-fable-5.1@low",
      "model_id": "anthropic/claude-fable-5.1",
      "model_label": "Claude Fable 5.1",
      "effort": "low",
      "label": "Claude Fable 5.1 \u00b7 Low",
      "short_label": "Claude Fable 5.1 Low",
      "score": 90.0,
      "passed": 81,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 31.45042404199997,
      "ci": [
        80.0,
        100.0
      ],
      "cost_known_attempts": 81,
      "tokens_known_attempts": 81,
      "cost_known_subtotal": 21.60685,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Stress": {
          "passed": 9,
          "attempts": 15,
          "rate": 60.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 10.623652208499845
      },
      "predictions": {
        "brier": {
          "metric": 0.06757547083333333,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.4642460796551948,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 81,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 81,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "anthropic/claude-fable-5.1@medium",
      "model_id": "anthropic/claude-fable-5.1",
      "model_label": "Claude Fable 5.1",
      "effort": "medium",
      "label": "Claude Fable 5.1 \u00b7 Medium",
      "short_label": "Claude Fable 5.1 Medium",
      "score": 92.22222222222223,
      "passed": 83,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 35.5328194579999,
      "ci": [
        83.33,
        100.0
      ],
      "cost_known_attempts": 83,
      "tokens_known_attempts": 83,
      "cost_known_subtotal": 26.45842,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 14,
          "attempts": 18,
          "rate": 77.77777777777777
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Stress": {
          "passed": 10,
          "attempts": 15,
          "rate": 66.66666666666667
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 10.269340604000142
      },
      "predictions": {
        "brier": {
          "metric": 0.07346390694444445,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.47840145767343634,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 83,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 83,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "anthropic/claude-fable-5.1@high",
      "model_id": "anthropic/claude-fable-5.1",
      "model_label": "Claude Fable 5.1",
      "effort": "high",
      "label": "Claude Fable 5.1 \u00b7 High",
      "short_label": "Claude Fable 5.1 High",
      "score": 93.33333333333333,
      "passed": 84,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 32.83930945850007,
      "ci": [
        85.56,
        100.0
      ],
      "cost_known_attempts": 84,
      "tokens_known_attempts": 84,
      "cost_known_subtotal": 30.57891,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Stress": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 10.752109353999955
      },
      "predictions": {
        "brier": {
          "metric": 0.07096350694444445,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.491908354574382,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 84,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 84,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "anthropic/claude-fable-5.1@xhigh",
      "model_id": "anthropic/claude-fable-5.1",
      "model_label": "Claude Fable 5.1",
      "effort": "xhigh",
      "label": "Claude Fable 5.1 \u00b7 XHigh",
      "short_label": "Claude Fable 5.1 XHigh",
      "score": 91.11111111111111,
      "passed": 82,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 34.99144839599985,
      "ci": [
        82.22,
        100.0
      ],
      "cost_known_attempts": 82,
      "tokens_known_attempts": 82,
      "cost_known_subtotal": 24.10059,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 13,
          "attempts": 18,
          "rate": 72.22222222222223
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Stress": {
          "passed": 10,
          "attempts": 15,
          "rate": 66.66666666666667
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 9.285665604000096
      },
      "predictions": {
        "brier": {
          "metric": 0.0745973111111111,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.46554381622550267,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 82,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 82,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "anthropic/claude-opus-5.5@low",
      "model_id": "anthropic/claude-opus-5.5",
      "model_label": "Claude Opus 5.5",
      "effort": "low",
      "label": "Claude Opus 5.5 \u00b7 Low",
      "short_label": "Claude Opus 5.5 Low",
      "score": 100.0,
      "passed": 90,
      "attempts": 90,
      "cost": 15.511800000000001,
      "tokens": 26053.233333333334,
      "latency": 23.368432583000015,
      "ci": [
        100.0,
        100.0
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 13.96062,
      "categories": {
        "Documents": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Stress": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 8.849430166999577
      },
      "predictions": {
        "brier": {
          "metric": 0.06873220416666667,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.43936722964527286,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 90,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 90,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "anthropic/claude-opus-5.5@medium",
      "model_id": "anthropic/claude-opus-5.5",
      "model_label": "Claude Opus 5.5",
      "effort": "medium",
      "label": "Claude Opus 5.5 \u00b7 Medium",
      "short_label": "Claude Opus 5.5 Medium",
      "score": 100.0,
      "passed": 90,
      "attempts": 90,
      "cost": 15.149977777777778,
      "tokens": 25257.966666666667,
      "latency": 21.93723185400013,
      "ci": [
        100.0,
        100.0
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 13.63498,
      "categories": {
        "Documents": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Stress": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 8.875758124999935
      },
      "predictions": {
        "brier": {
          "metric": 0.07032890972222222,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.42836716717347206,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 90,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 90,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "anthropic/claude-opus-5.5@high",
      "model_id": "anthropic/claude-opus-5.5",
      "model_label": "Claude Opus 5.5",
      "effort": "high",
      "label": "Claude Opus 5.5 \u00b7 High",
      "short_label": "Claude Opus 5.5 High",
      "score": 98.88888888888889,
      "passed": 89,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 22.174622667000047,
      "ci": [
        96.67,
        100.0
      ],
      "cost_known_attempts": 89,
      "tokens_known_attempts": 89,
      "cost_known_subtotal": 13.13436,
      "categories": {
        "Documents": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Stress": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 9.086503749500146
      },
      "predictions": {
        "brier": {
          "metric": 0.06900902777777777,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.4405147802637737,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 89,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 89,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "anthropic/claude-opus-5.5@xhigh",
      "model_id": "anthropic/claude-opus-5.5",
      "model_label": "Claude Opus 5.5",
      "effort": "xhigh",
      "label": "Claude Opus 5.5 \u00b7 XHigh",
      "short_label": "Claude Opus 5.5 XHigh",
      "score": 98.88888888888889,
      "passed": 89,
      "attempts": 90,
      "cost": 14.27929438202247,
      "tokens": 23130.78888888889,
      "latency": 22.79177179200016,
      "ci": [
        96.67,
        100.0
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 12.708572,
      "categories": {
        "Documents": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Stress": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 9.342110729500186
      },
      "predictions": {
        "brier": {
          "metric": 0.0668275125,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.4265938066349015,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 89,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 89,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "anthropic/claude-opus-5.5@max",
      "model_id": "anthropic/claude-opus-5.5",
      "model_label": "Claude Opus 5.5",
      "effort": "max",
      "label": "Claude Opus 5.5 \u00b7 Max",
      "short_label": "Claude Opus 5.5 Max",
      "score": 100.0,
      "passed": 90,
      "attempts": 90,
      "cost": 17.420235555555557,
      "tokens": 29689.033333333333,
      "latency": 23.272986312499736,
      "ci": [
        100.0,
        100.0
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 15.678212,
      "categories": {
        "Documents": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Stress": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 9.462106478999484
      },
      "predictions": {
        "brier": {
          "metric": 0.06758623888888889,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.4289136924214509,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 90,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 90,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "anthropic/claude-opus-5@low",
      "model_id": "anthropic/claude-opus-5",
      "model_label": "Claude Opus 5",
      "effort": "low",
      "label": "Claude Opus 5 \u00b7 Low",
      "short_label": "Claude Opus 5 Low",
      "score": 93.33333333333333,
      "passed": 84,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 30.25929422949995,
      "ci": [
        85.56,
        98.89
      ],
      "cost_known_attempts": 85,
      "tokens_known_attempts": 85,
      "cost_known_subtotal": 19.21472,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        },
        "Knowledge": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Stress": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 15.439800604000222
      },
      "predictions": {
        "brier": {
          "metric": 0.06933992916666666,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.47951489160253963,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 74,
      "parser_changed_delivery": 10,
      "parser_delivery_passed": 84,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "anthropic/claude-opus-5@medium",
      "model_id": "anthropic/claude-opus-5",
      "model_label": "Claude Opus 5",
      "effort": "medium",
      "label": "Claude Opus 5 \u00b7 Medium",
      "short_label": "Claude Opus 5 Medium",
      "score": 94.44444444444444,
      "passed": 85,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 32.75709225000022,
      "ci": [
        86.67,
        100.0
      ],
      "cost_known_attempts": 87,
      "tokens_known_attempts": 87,
      "cost_known_subtotal": 21.31135,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Apps": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Stress": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 14.210771041500383
      },
      "predictions": {
        "brier": {
          "metric": 0.06830246944444444,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.4921189452826235,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 73,
      "parser_changed_delivery": 12,
      "parser_delivery_passed": 85,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "anthropic/claude-opus-5@high",
      "model_id": "anthropic/claude-opus-5",
      "model_label": "Claude Opus 5",
      "effort": "high",
      "label": "Claude Opus 5 \u00b7 High",
      "short_label": "Claude Opus 5 High",
      "score": 93.33333333333333,
      "passed": 84,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 31.231335562500405,
      "ci": [
        86.67,
        98.89
      ],
      "cost_known_attempts": 86,
      "tokens_known_attempts": 86,
      "cost_known_subtotal": 22.08886,
      "categories": {
        "Documents": {
          "passed": 14,
          "attempts": 18,
          "rate": 77.77777777777777
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Apps": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Stress": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 13.045234770999988
      },
      "predictions": {
        "brier": {
          "metric": 0.06903178333333333,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.4823239462219235,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 72,
      "parser_changed_delivery": 12,
      "parser_delivery_passed": 84,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 1
    },
    {
      "id": "anthropic/claude-opus-5@xhigh",
      "model_id": "anthropic/claude-opus-5",
      "model_label": "Claude Opus 5",
      "effort": "xhigh",
      "label": "Claude Opus 5 \u00b7 XHigh",
      "short_label": "Claude Opus 5 XHigh",
      "score": 92.22222222222223,
      "passed": 83,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 29.6669030830001,
      "ci": [
        84.44,
        98.89
      ],
      "cost_known_attempts": 86,
      "tokens_known_attempts": 86,
      "cost_known_subtotal": 23.823045,
      "categories": {
        "Documents": {
          "passed": 14,
          "attempts": 18,
          "rate": 77.77777777777777
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        },
        "Apps": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Stress": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 12.780636771000108
      },
      "predictions": {
        "brier": {
          "metric": 0.06852921388888888,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.4828095707063377,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 72,
      "parser_changed_delivery": 11,
      "parser_delivery_passed": 83,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 1
    },
    {
      "id": "anthropic/claude-opus-5@max",
      "model_id": "anthropic/claude-opus-5",
      "model_label": "Claude Opus 5",
      "effort": "max",
      "label": "Claude Opus 5 \u00b7 Max",
      "short_label": "Claude Opus 5 Max",
      "score": 95.55555555555556,
      "passed": 86,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 31.296417937500053,
      "ci": [
        88.89,
        100.0
      ],
      "cost_known_attempts": 86,
      "tokens_known_attempts": 86,
      "cost_known_subtotal": 20.7155,
      "categories": {
        "Documents": {
          "passed": 14,
          "attempts": 18,
          "rate": 77.77777777777777
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Stress": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 13.423790750000045
      },
      "predictions": {
        "brier": {
          "metric": 0.07035066527777778,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.4770625552474852,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 74,
      "parser_changed_delivery": 12,
      "parser_delivery_passed": 86,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 1
    },
    {
      "id": "anthropic/claude-haiku-4.5@none",
      "model_id": "anthropic/claude-haiku-4.5",
      "model_label": "Claude Haiku 4.5",
      "effort": "none",
      "label": "Claude Haiku 4.5 \u00b7 Off",
      "short_label": "Claude Haiku 4.5 Off",
      "score": 31.11111111111111,
      "passed": 28,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 25.138010687999994,
      "ci": [
        21.11,
        41.11
      ],
      "cost_known_attempts": 89,
      "tokens_known_attempts": 89,
      "cost_known_subtotal": 10.78739,
      "categories": {
        "Documents": {
          "passed": 7,
          "attempts": 18,
          "rate": 38.888888888888886
        },
        "Takeoffs": {
          "passed": 7,
          "attempts": 18,
          "rate": 38.888888888888886
        },
        "Analysis": {
          "passed": 3,
          "attempts": 18,
          "rate": 16.666666666666668
        },
        "Knowledge": {
          "passed": 10,
          "attempts": 18,
          "rate": 55.55555555555556
        },
        "Apps": {
          "passed": 1,
          "attempts": 18,
          "rate": 5.555555555555555
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 8,
          "attempts": 15,
          "rate": 53.333333333333336
        },
        "Multi-step": {
          "passed": 7,
          "attempts": 15,
          "rate": 46.666666666666664
        },
        "Complex": {
          "passed": 6,
          "attempts": 15,
          "rate": 40.0
        },
        "Expert": {
          "passed": 5,
          "attempts": 15,
          "rate": 33.333333333333336
        },
        "Frontier": {
          "passed": 2,
          "attempts": 15,
          "rate": 13.333333333333334
        },
        "Stress": {
          "passed": 0,
          "attempts": 15,
          "rate": 0.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 3.8575351454997433
      },
      "predictions": {
        "brier": {
          "metric": 0.23789199933333333,
          "baseline_matched": 0.07330511232398954,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 5,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.38947483612712774,
          "baseline_matched": 0.3894748361271277,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 10,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 28,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 28,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 1
    },
    {
      "id": "anthropic/claude-haiku-4.5@medium",
      "model_id": "anthropic/claude-haiku-4.5",
      "model_label": "Claude Haiku 4.5",
      "effort": "medium",
      "label": "Claude Haiku 4.5 \u00b7 Medium",
      "short_label": "Claude Haiku 4.5 Medium",
      "score": 32.22222222222222,
      "passed": 29,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 29.326590417,
      "ci": [
        23.33,
        42.22
      ],
      "cost_known_attempts": 89,
      "tokens_known_attempts": 89,
      "cost_known_subtotal": 10.799374,
      "categories": {
        "Documents": {
          "passed": 9,
          "attempts": 18,
          "rate": 50.0
        },
        "Takeoffs": {
          "passed": 6,
          "attempts": 18,
          "rate": 33.333333333333336
        },
        "Analysis": {
          "passed": 3,
          "attempts": 18,
          "rate": 16.666666666666668
        },
        "Knowledge": {
          "passed": 10,
          "attempts": 18,
          "rate": 55.55555555555556
        },
        "Apps": {
          "passed": 1,
          "attempts": 18,
          "rate": 5.555555555555555
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 6,
          "attempts": 15,
          "rate": 40.0
        },
        "Multi-step": {
          "passed": 9,
          "attempts": 15,
          "rate": 60.0
        },
        "Complex": {
          "passed": 5,
          "attempts": 15,
          "rate": 33.333333333333336
        },
        "Expert": {
          "passed": 4,
          "attempts": 15,
          "rate": 26.666666666666668
        },
        "Frontier": {
          "passed": 3,
          "attempts": 15,
          "rate": 20.0
        },
        "Stress": {
          "passed": 2,
          "attempts": 15,
          "rate": 13.333333333333334
        }
      },
      "judgment": {
        "correct": 11,
        "attempts": 12,
        "correct_rate": 91.66666666666667,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 6.236981334000127
      },
      "predictions": {
        "brier": {
          "metric": 0.1770731412,
          "baseline_matched": 0.07421171351471649,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 5,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.307874466552738,
          "baseline_matched": 0.3058889069859509,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 8,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 29,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 29,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 1
    },
    {
      "id": "anthropic/claude-haiku-4.5@high",
      "model_id": "anthropic/claude-haiku-4.5",
      "model_label": "Claude Haiku 4.5",
      "effort": "high",
      "label": "Claude Haiku 4.5 \u00b7 High",
      "short_label": "Claude Haiku 4.5 High",
      "score": 37.77777777777778,
      "passed": 34,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 26.771920000000044,
      "ci": [
        26.67,
        47.78
      ],
      "cost_known_attempts": 89,
      "tokens_known_attempts": 89,
      "cost_known_subtotal": 11.215735,
      "categories": {
        "Documents": {
          "passed": 11,
          "attempts": 18,
          "rate": 61.111111111111114
        },
        "Takeoffs": {
          "passed": 6,
          "attempts": 18,
          "rate": 33.333333333333336
        },
        "Analysis": {
          "passed": 5,
          "attempts": 18,
          "rate": 27.77777777777778
        },
        "Knowledge": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Apps": {
          "passed": 0,
          "attempts": 18,
          "rate": 0.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 9,
          "attempts": 15,
          "rate": 60.0
        },
        "Multi-step": {
          "passed": 8,
          "attempts": 15,
          "rate": 53.333333333333336
        },
        "Complex": {
          "passed": 8,
          "attempts": 15,
          "rate": 53.333333333333336
        },
        "Expert": {
          "passed": 6,
          "attempts": 15,
          "rate": 40.0
        },
        "Frontier": {
          "passed": 2,
          "attempts": 15,
          "rate": 13.333333333333334
        },
        "Stress": {
          "passed": 1,
          "attempts": 15,
          "rate": 6.666666666666667
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 4.222743020499941
      },
      "predictions": {
        "brier": {
          "metric": 0.156973,
          "baseline_matched": 0.052245502612021585,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 7,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.3364794370238455,
          "baseline_matched": 0.3364307159519818,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 10,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 34,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 34,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 1
    },
    {
      "id": "spacexai/grok-4.7@fixed",
      "model_id": "spacexai/grok-4.7",
      "model_label": "Grok 4.7",
      "effort": "fixed",
      "label": "Grok 4.7 \u00b7 Fixed",
      "short_label": "Grok 4.7 Fixed",
      "score": 68.88888888888889,
      "passed": 62,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 48.801221124999785,
      "ci": [
        57.78,
        78.89
      ],
      "cost_known_attempts": 67,
      "tokens_known_attempts": 67,
      "cost_known_subtotal": 2.7320256,
      "categories": {
        "Documents": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 13,
          "attempts": 18,
          "rate": 72.22222222222223
        },
        "Knowledge": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Apps": {
          "passed": 2,
          "attempts": 18,
          "rate": 11.11111111111111
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Multi-step": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Complex": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Expert": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Frontier": {
          "passed": 6,
          "attempts": 15,
          "rate": 40.0
        },
        "Stress": {
          "passed": 6,
          "attempts": 15,
          "rate": 40.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 9.843619124999968
      },
      "predictions": {
        "brier": {
          "metric": null,
          "baseline_matched": null,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 0,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.37725912746250945,
          "baseline_matched": 0.3772591274625094,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 11,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 62,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 62,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 1
    },
    {
      "id": "spacexai/grok-build-0.1@fixed",
      "model_id": "spacexai/grok-build-0.1",
      "model_label": "Grok Build 0.1",
      "effort": "fixed",
      "label": "Grok Build 0.1 \u00b7 Fixed",
      "short_label": "Grok Build 0.1 Fixed",
      "score": 44.44444444444444,
      "passed": 40,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 46.18986424999963,
      "ci": [
        32.22,
        56.67
      ],
      "cost_known_attempts": 74,
      "tokens_known_attempts": 74,
      "cost_known_subtotal": 1.57672,
      "categories": {
        "Documents": {
          "passed": 9,
          "attempts": 18,
          "rate": 50.0
        },
        "Takeoffs": {
          "passed": 6,
          "attempts": 18,
          "rate": 33.333333333333336
        },
        "Analysis": {
          "passed": 9,
          "attempts": 18,
          "rate": 50.0
        },
        "Knowledge": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Apps": {
          "passed": 4,
          "attempts": 18,
          "rate": 22.22222222222222
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 7,
          "attempts": 15,
          "rate": 46.666666666666664
        },
        "Multi-step": {
          "passed": 6,
          "attempts": 15,
          "rate": 40.0
        },
        "Complex": {
          "passed": 8,
          "attempts": 15,
          "rate": 53.333333333333336
        },
        "Expert": {
          "passed": 7,
          "attempts": 15,
          "rate": 46.666666666666664
        },
        "Frontier": {
          "passed": 5,
          "attempts": 15,
          "rate": 33.333333333333336
        },
        "Stress": {
          "passed": 7,
          "attempts": 15,
          "rate": 46.666666666666664
        }
      },
      "judgment": {
        "correct": 10,
        "attempts": 12,
        "correct_rate": 83.33333333333333,
        "false_stops": 6,
        "solvable_attempts": 90,
        "false_stop_rate": 6.666666666666667,
        "median_correct_stop_s": 11.952241853999787
      },
      "predictions": {
        "brier": {
          "metric": 0.18430540666666667,
          "baseline_matched": 0.07511831470544343,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 5,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.3548349036483021,
          "baseline_matched": 0.3548349036483021,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 11,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 40,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 40,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 1
    },
    {
      "id": "spacexai/grok-4.20-reasoning@none",
      "model_id": "spacexai/grok-4.20-reasoning",
      "model_label": "Grok 4.20",
      "effort": "none",
      "label": "Grok 4.20 \u00b7 Off",
      "short_label": "Grok 4.20 Off",
      "score": 34.44444444444444,
      "passed": 31,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 11.096592374999076,
      "ci": [
        23.33,
        45.58
      ],
      "cost_known_attempts": 80,
      "tokens_known_attempts": 80,
      "cost_known_subtotal": 1.9141602,
      "categories": {
        "Documents": {
          "passed": 11,
          "attempts": 18,
          "rate": 61.111111111111114
        },
        "Takeoffs": {
          "passed": 5,
          "attempts": 18,
          "rate": 27.77777777777778
        },
        "Analysis": {
          "passed": 5,
          "attempts": 18,
          "rate": 27.77777777777778
        },
        "Knowledge": {
          "passed": 8,
          "attempts": 18,
          "rate": 44.44444444444444
        },
        "Apps": {
          "passed": 2,
          "attempts": 18,
          "rate": 11.11111111111111
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 7,
          "attempts": 15,
          "rate": 46.666666666666664
        },
        "Multi-step": {
          "passed": 6,
          "attempts": 15,
          "rate": 40.0
        },
        "Complex": {
          "passed": 6,
          "attempts": 15,
          "rate": 40.0
        },
        "Expert": {
          "passed": 8,
          "attempts": 15,
          "rate": 53.333333333333336
        },
        "Frontier": {
          "passed": 2,
          "attempts": 15,
          "rate": 13.333333333333334
        },
        "Stress": {
          "passed": 2,
          "attempts": 15,
          "rate": 13.333333333333334
        }
      },
      "judgment": {
        "correct": 5,
        "attempts": 12,
        "correct_rate": 41.666666666666664,
        "false_stops": 1,
        "solvable_attempts": 90,
        "false_stop_rate": 1.1111111111111112,
        "median_correct_stop_s": 1.2093355000000447
      },
      "predictions": {
        "brier": {
          "metric": 0.1657752359,
          "baseline_matched": 0.04310500624877206,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 5,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.2755607886678829,
          "baseline_matched": 0.2755607886678828,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 6,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 31,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 31,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 1
    },
    {
      "id": "spacexai/grok-4.20-reasoning@high",
      "model_id": "spacexai/grok-4.20-reasoning",
      "model_label": "Grok 4.20",
      "effort": "high",
      "label": "Grok 4.20 \u00b7 High",
      "short_label": "Grok 4.20 High",
      "score": 75.55555555555556,
      "passed": 68,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 38.78002095850039,
      "ci": [
        64.44,
        86.67
      ],
      "cost_known_attempts": 85,
      "tokens_known_attempts": 85,
      "cost_known_subtotal": 2.9347143,
      "categories": {
        "Documents": {
          "passed": 13,
          "attempts": 18,
          "rate": 72.22222222222223
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 13,
          "attempts": 18,
          "rate": 72.22222222222223
        },
        "Knowledge": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Apps": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Multi-step": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        },
        "Complex": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Expert": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Frontier": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        },
        "Stress": {
          "passed": 9,
          "attempts": 15,
          "rate": 60.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 7,
        "solvable_attempts": 90,
        "false_stop_rate": 7.777777777777778,
        "median_correct_stop_s": 8.05594635450002
      },
      "predictions": {
        "brier": {
          "metric": 0.15786275676666667,
          "baseline_matched": 0.0582050592863808,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 5,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.3556956978124052,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 67,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 67,
      "semantic_changed_delivery": 1,
      "infrastructure_interruptions": 1
    },
    {
      "id": "google/gemini-3.8-flash@low",
      "model_id": "google/gemini-3.8-flash",
      "model_label": "Gemini 3.8 Flash",
      "effort": "low",
      "label": "Gemini 3.8 Flash \u00b7 Low",
      "short_label": "Gemini 3.8 Flash Low",
      "score": 83.33333333333333,
      "passed": 75,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 23.053716583999805,
      "ci": [
        72.22,
        93.33
      ],
      "cost_known_attempts": 88,
      "tokens_known_attempts": 88,
      "cost_known_subtotal": 3.204907875,
      "categories": {
        "Documents": {
          "passed": 13,
          "attempts": 18,
          "rate": 72.22222222222223
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 14,
          "attempts": 18,
          "rate": 77.77777777777777
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Multi-step": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 10,
          "attempts": 15,
          "rate": 66.66666666666667
        },
        "Stress": {
          "passed": 7,
          "attempts": 15,
          "rate": 46.666666666666664
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 5.945441978999879
      },
      "predictions": {
        "brier": {
          "metric": 0.06814781618055556,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.6214327990521099,
          "baseline_matched": 0.35918383316171026,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 9,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 75,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 75,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 1
    },
    {
      "id": "google/gemini-3.8-flash@medium",
      "model_id": "google/gemini-3.8-flash",
      "model_label": "Gemini 3.8 Flash",
      "effort": "medium",
      "label": "Gemini 3.8 Flash \u00b7 Medium",
      "short_label": "Gemini 3.8 Flash Medium",
      "score": 57.77777777777778,
      "passed": 52,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 36.82981270800001,
      "ci": [
        47.78,
        67.78
      ],
      "cost_known_attempts": 84,
      "tokens_known_attempts": 84,
      "cost_known_subtotal": 3.972715575,
      "categories": {
        "Documents": {
          "passed": 11,
          "attempts": 18,
          "rate": 61.111111111111114
        },
        "Takeoffs": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        },
        "Analysis": {
          "passed": 7,
          "attempts": 18,
          "rate": 38.888888888888886
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 0,
          "attempts": 18,
          "rate": 0.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Multi-step": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        },
        "Complex": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        },
        "Expert": {
          "passed": 8,
          "attempts": 15,
          "rate": 53.333333333333336
        },
        "Frontier": {
          "passed": 5,
          "attempts": 15,
          "rate": 33.333333333333336
        },
        "Stress": {
          "passed": 5,
          "attempts": 15,
          "rate": 33.333333333333336
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 13.609862521000089
      },
      "predictions": {
        "brier": {
          "metric": null,
          "baseline_matched": null,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 0,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.4603385750453503,
          "baseline_matched": 0.3771531279066897,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 10,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 52,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 52,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "google/gemini-3.8-flash@high",
      "model_id": "google/gemini-3.8-flash",
      "model_label": "Gemini 3.8 Flash",
      "effort": "high",
      "label": "Gemini 3.8 Flash \u00b7 High",
      "short_label": "Gemini 3.8 Flash High",
      "score": 47.77777777777778,
      "passed": 43,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 78.05775924999999,
      "ci": [
        35.56,
        60.0
      ],
      "cost_known_attempts": 87,
      "tokens_known_attempts": 87,
      "cost_known_subtotal": 6.153970575,
      "categories": {
        "Documents": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Takeoffs": {
          "passed": 7,
          "attempts": 18,
          "rate": 38.888888888888886
        },
        "Analysis": {
          "passed": 6,
          "attempts": 18,
          "rate": 33.333333333333336
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 0,
          "attempts": 18,
          "rate": 0.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Multi-step": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        },
        "Complex": {
          "passed": 8,
          "attempts": 15,
          "rate": 53.333333333333336
        },
        "Expert": {
          "passed": 6,
          "attempts": 15,
          "rate": 40.0
        },
        "Frontier": {
          "passed": 3,
          "attempts": 15,
          "rate": 20.0
        },
        "Stress": {
          "passed": 3,
          "attempts": 15,
          "rate": 20.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 22.43144487499993
      },
      "predictions": {
        "brier": {
          "metric": null,
          "baseline_matched": null,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 0,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.27653061224489794,
          "baseline_matched": 0.2551020408163266,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 2,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 43,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 43,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "google/gemini-3.5-flash-lite@minimal",
      "model_id": "google/gemini-3.5-flash-lite",
      "model_label": "Gemini 3.5 Flash Lite",
      "effort": "minimal",
      "label": "Gemini 3.5 Flash Lite \u00b7 Minimal",
      "short_label": "Gemini 3.5 Flash Lite Minimal",
      "score": 46.666666666666664,
      "passed": 42,
      "attempts": 90,
      "cost": 4.521153047619047,
      "tokens": 47104.92222222222,
      "latency": 13.456935958499233,
      "ci": [
        31.11,
        61.11
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 1.8988842799999999,
      "categories": {
        "Documents": {
          "passed": 5,
          "attempts": 18,
          "rate": 27.77777777777778
        },
        "Takeoffs": {
          "passed": 8,
          "attempts": 18,
          "rate": 44.44444444444444
        },
        "Analysis": {
          "passed": 8,
          "attempts": 18,
          "rate": 44.44444444444444
        },
        "Knowledge": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Apps": {
          "passed": 9,
          "attempts": 18,
          "rate": 50.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        },
        "Multi-step": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Complex": {
          "passed": 6,
          "attempts": 15,
          "rate": 40.0
        },
        "Expert": {
          "passed": 6,
          "attempts": 15,
          "rate": 40.0
        },
        "Frontier": {
          "passed": 3,
          "attempts": 15,
          "rate": 20.0
        },
        "Stress": {
          "passed": 3,
          "attempts": 15,
          "rate": 20.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 3.180724750500173
      },
      "predictions": {
        "brier": {
          "metric": 0.3246552396458333,
          "baseline_matched": 0.07769393814652457,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 8,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.5911144886418579,
          "baseline_matched": 0.37222309439832335,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 8,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 42,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 42,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "google/gemini-3.5-flash-lite@low",
      "model_id": "google/gemini-3.5-flash-lite",
      "model_label": "Gemini 3.5 Flash Lite",
      "effort": "low",
      "label": "Gemini 3.5 Flash Lite \u00b7 Low",
      "short_label": "Gemini 3.5 Flash Lite Low",
      "score": 57.77777777777778,
      "passed": 52,
      "attempts": 90,
      "cost": 3.2246806538461543,
      "tokens": 47659.02222222222,
      "latency": 13.254216583500005,
      "ci": [
        43.33,
        72.22
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 1.67683394,
      "categories": {
        "Documents": {
          "passed": 9,
          "attempts": 18,
          "rate": 50.0
        },
        "Takeoffs": {
          "passed": 11,
          "attempts": 18,
          "rate": 61.111111111111114
        },
        "Analysis": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Knowledge": {
          "passed": 14,
          "attempts": 18,
          "rate": 77.77777777777777
        },
        "Apps": {
          "passed": 6,
          "attempts": 18,
          "rate": 33.333333333333336
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Multi-step": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Complex": {
          "passed": 10,
          "attempts": 15,
          "rate": 66.66666666666667
        },
        "Expert": {
          "passed": 10,
          "attempts": 15,
          "rate": 66.66666666666667
        },
        "Frontier": {
          "passed": 5,
          "attempts": 15,
          "rate": 33.333333333333336
        },
        "Stress": {
          "passed": 3,
          "attempts": 15,
          "rate": 20.0
        }
      },
      "judgment": {
        "correct": 10,
        "attempts": 12,
        "correct_rate": 83.33333333333333,
        "false_stops": 2,
        "solvable_attempts": 90,
        "false_stop_rate": 2.2222222222222223,
        "median_correct_stop_s": 3.732497165999841
      },
      "predictions": {
        "brier": {
          "metric": 0.235208667,
          "baseline_matched": 0.06472526368067763,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 11,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.6344295053024949,
          "baseline_matched": 0.36603645657597295,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 11,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 52,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 52,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "google/gemini-3.5-flash-lite@medium",
      "model_id": "google/gemini-3.5-flash-lite",
      "model_label": "Gemini 3.5 Flash Lite",
      "effort": "medium",
      "label": "Gemini 3.5 Flash Lite \u00b7 Medium",
      "short_label": "Gemini 3.5 Flash Lite Medium",
      "score": 63.333333333333336,
      "passed": 57,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 26.653550374999643,
      "ci": [
        51.11,
        75.56
      ],
      "cost_known_attempts": 87,
      "tokens_known_attempts": 87,
      "cost_known_subtotal": 2.23744287,
      "categories": {
        "Documents": {
          "passed": 13,
          "attempts": 18,
          "rate": 72.22222222222223
        },
        "Takeoffs": {
          "passed": 11,
          "attempts": 18,
          "rate": 61.111111111111114
        },
        "Analysis": {
          "passed": 9,
          "attempts": 18,
          "rate": 50.0
        },
        "Knowledge": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        },
        "Apps": {
          "passed": 8,
          "attempts": 18,
          "rate": 44.44444444444444
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Complex": {
          "passed": 10,
          "attempts": 15,
          "rate": 66.66666666666667
        },
        "Expert": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        },
        "Frontier": {
          "passed": 6,
          "attempts": 15,
          "rate": 40.0
        },
        "Stress": {
          "passed": 2,
          "attempts": 15,
          "rate": 13.333333333333334
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 13.447978395999991
      },
      "predictions": {
        "brier": {
          "metric": 0.39652299585714285,
          "baseline_matched": 0.06958445594213085,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 7,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.5410784573842766,
          "baseline_matched": 0.3360809818088684,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 7,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 57,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 57,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "google/gemini-3.5-flash-lite@high",
      "model_id": "google/gemini-3.5-flash-lite",
      "model_label": "Gemini 3.5 Flash Lite",
      "effort": "high",
      "label": "Gemini 3.5 Flash Lite \u00b7 High",
      "short_label": "Gemini 3.5 Flash Lite High",
      "score": 46.666666666666664,
      "passed": 42,
      "attempts": 90,
      "cost": 6.485538619047619,
      "tokens": 66165.82222222222,
      "latency": 30.37339079199964,
      "ci": [
        35.56,
        58.89
      ],
      "cost_known_attempts": 90,
      "tokens_known_attempts": 90,
      "cost_known_subtotal": 2.72392622,
      "categories": {
        "Documents": {
          "passed": 11,
          "attempts": 18,
          "rate": 61.111111111111114
        },
        "Takeoffs": {
          "passed": 8,
          "attempts": 18,
          "rate": 44.44444444444444
        },
        "Analysis": {
          "passed": 7,
          "attempts": 18,
          "rate": 38.888888888888886
        },
        "Knowledge": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Apps": {
          "passed": 1,
          "attempts": 18,
          "rate": 5.555555555555555
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 10,
          "attempts": 15,
          "rate": 66.66666666666667
        },
        "Multi-step": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Complex": {
          "passed": 9,
          "attempts": 15,
          "rate": 60.0
        },
        "Expert": {
          "passed": 8,
          "attempts": 15,
          "rate": 53.333333333333336
        },
        "Frontier": {
          "passed": 2,
          "attempts": 15,
          "rate": 13.333333333333334
        },
        "Stress": {
          "passed": 1,
          "attempts": 15,
          "rate": 6.666666666666667
        }
      },
      "judgment": {
        "correct": 11,
        "attempts": 12,
        "correct_rate": 91.66666666666667,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 13.489078041999601
      },
      "predictions": {
        "brier": {
          "metric": 0.43609475,
          "baseline_matched": 0.05442704243560219,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 1,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.3380102040816326,
          "baseline_matched": 0.2551020408163266,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 2,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 42,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 42,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "google/gemini-3.1-pro-preview@low",
      "model_id": "google/gemini-3.1-pro-preview",
      "model_label": "Gemini 3.1 Pro Preview",
      "effort": "low",
      "label": "Gemini 3.1 Pro Preview \u00b7 Low",
      "short_label": "Gemini 3.1 Pro Preview Low",
      "score": 90.0,
      "passed": 81,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 24.53602037499938,
      "ci": [
        80.0,
        100.0
      ],
      "cost_known_attempts": 81,
      "tokens_known_attempts": 81,
      "cost_known_subtotal": 3.8767618,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Stress": {
          "passed": 9,
          "attempts": 15,
          "rate": 60.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 8.026528916500043
      },
      "predictions": {
        "brier": {
          "metric": 0.1295629178238513,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.6830180335390641,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 81,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 81,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "google/gemini-3.1-pro-preview@medium",
      "model_id": "google/gemini-3.1-pro-preview",
      "model_label": "Gemini 3.1 Pro Preview",
      "effort": "medium",
      "label": "Gemini 3.1 Pro Preview \u00b7 Medium",
      "short_label": "Gemini 3.1 Pro Preview Medium",
      "score": 88.88888888888889,
      "passed": 80,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 35.64942943750019,
      "ci": [
        78.89,
        97.78
      ],
      "cost_known_attempts": 81,
      "tokens_known_attempts": 81,
      "cost_known_subtotal": 5.3684352,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Analysis": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Multi-step": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Complex": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Stress": {
          "passed": 8,
          "attempts": 15,
          "rate": 53.333333333333336
        }
      },
      "judgment": {
        "correct": 10,
        "attempts": 12,
        "correct_rate": 83.33333333333333,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 10.537503353999927
      },
      "predictions": {
        "brier": {
          "metric": 0.1451453887870496,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.6766801542684024,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 80,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 80,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "google/gemini-3.1-pro-preview@high",
      "model_id": "google/gemini-3.1-pro-preview",
      "model_label": "Gemini 3.1 Pro Preview",
      "effort": "high",
      "label": "Gemini 3.1 Pro Preview \u00b7 High",
      "short_label": "Gemini 3.1 Pro Preview High",
      "score": 78.88888888888889,
      "passed": 71,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 66.57771304200007,
      "ci": [
        66.67,
        90.0
      ],
      "cost_known_attempts": 81,
      "tokens_known_attempts": 81,
      "cost_known_subtotal": 10.8935636,
      "categories": {
        "Documents": {
          "passed": 14,
          "attempts": 18,
          "rate": 77.77777777777777
        },
        "Takeoffs": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Analysis": {
          "passed": 11,
          "attempts": 18,
          "rate": 61.111111111111114
        },
        "Knowledge": {
          "passed": 18,
          "attempts": 18,
          "rate": 100.0
        },
        "Apps": {
          "passed": 10,
          "attempts": 18,
          "rate": 55.55555555555556
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        },
        "Multi-step": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Complex": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Expert": {
          "passed": 15,
          "attempts": 15,
          "rate": 100.0
        },
        "Frontier": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        },
        "Stress": {
          "passed": 9,
          "attempts": 15,
          "rate": 60.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 12.153207958499902
      },
      "predictions": {
        "brier": {
          "metric": 0.10605874551718182,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.617040326301264,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 71,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 71,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    },
    {
      "id": "zai/glm-5.3@low",
      "model_id": "zai/glm-5.3",
      "model_label": "GLM 5.3",
      "effort": "low",
      "label": "GLM 5.3 \u00b7 Low",
      "short_label": "GLM 5.3 Low",
      "score": 63.333333333333336,
      "passed": 57,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 19.655861959000116,
      "ci": [
        52.22,
        74.44
      ],
      "cost_known_attempts": 87,
      "tokens_known_attempts": 87,
      "cost_known_subtotal": 3.28726476,
      "categories": {
        "Documents": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Takeoffs": {
          "passed": 7,
          "attempts": 18,
          "rate": 38.888888888888886
        },
        "Analysis": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Knowledge": {
          "passed": 14,
          "attempts": 18,
          "rate": 77.77777777777777
        },
        "Apps": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Multi-step": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        },
        "Complex": {
          "passed": 9,
          "attempts": 15,
          "rate": 60.0
        },
        "Expert": {
          "passed": 10,
          "attempts": 15,
          "rate": 66.66666666666667
        },
        "Frontier": {
          "passed": 9,
          "attempts": 15,
          "rate": 60.0
        },
        "Stress": {
          "passed": 6,
          "attempts": 15,
          "rate": 40.0
        }
      },
      "judgment": {
        "correct": 11,
        "attempts": 12,
        "correct_rate": 91.66666666666667,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 3.024182625000365
      },
      "predictions": {
        "brier": {
          "metric": 0.15270434305555555,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.6267189557636639,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 55,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 55,
      "semantic_changed_delivery": 2,
      "infrastructure_interruptions": 0
    },
    {
      "id": "zai/glm-5.3@high",
      "model_id": "zai/glm-5.3",
      "model_label": "GLM 5.3",
      "effort": "high",
      "label": "GLM 5.3 \u00b7 High",
      "short_label": "GLM 5.3 High",
      "score": 74.44444444444444,
      "passed": 67,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 53.02552470899932,
      "ci": [
        63.33,
        85.56
      ],
      "cost_known_attempts": 82,
      "tokens_known_attempts": 82,
      "cost_known_subtotal": 3.08588864,
      "categories": {
        "Documents": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        },
        "Takeoffs": {
          "passed": 8,
          "attempts": 18,
          "rate": 44.44444444444444
        },
        "Analysis": {
          "passed": 13,
          "attempts": 18,
          "rate": 72.22222222222223
        },
        "Knowledge": {
          "passed": 16,
          "attempts": 18,
          "rate": 88.88888888888889
        },
        "Apps": {
          "passed": 15,
          "attempts": 18,
          "rate": 83.33333333333333
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Multi-step": {
          "passed": 14,
          "attempts": 15,
          "rate": 93.33333333333333
        },
        "Complex": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Expert": {
          "passed": 13,
          "attempts": 15,
          "rate": 86.66666666666667
        },
        "Frontier": {
          "passed": 8,
          "attempts": 15,
          "rate": 53.333333333333336
        },
        "Stress": {
          "passed": 7,
          "attempts": 15,
          "rate": 46.666666666666664
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 3.9176439375001935
      },
      "predictions": {
        "brier": {
          "metric": 0.07636218680555555,
          "baseline_matched": 0.06768977925381477,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.5293912471811874,
          "baseline_matched": 0.3567919219293358,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 12,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 66,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 66,
      "semantic_changed_delivery": 1,
      "infrastructure_interruptions": 0
    },
    {
      "id": "zai/glm-5.3@max",
      "model_id": "zai/glm-5.3",
      "model_label": "GLM 5.3",
      "effort": "max",
      "label": "GLM 5.3 \u00b7 Max",
      "short_label": "GLM 5.3 Max",
      "score": 67.77777777777777,
      "passed": 61,
      "attempts": 90,
      "cost": null,
      "tokens": null,
      "latency": 67.30587366699986,
      "ci": [
        56.67,
        78.89
      ],
      "cost_known_attempts": 70,
      "tokens_known_attempts": 70,
      "cost_known_subtotal": 2.93505636,
      "categories": {
        "Documents": {
          "passed": 14,
          "attempts": 18,
          "rate": 77.77777777777777
        },
        "Takeoffs": {
          "passed": 13,
          "attempts": 18,
          "rate": 72.22222222222223
        },
        "Analysis": {
          "passed": 12,
          "attempts": 18,
          "rate": 66.66666666666667
        },
        "Knowledge": {
          "passed": 17,
          "attempts": 18,
          "rate": 94.44444444444444
        },
        "Apps": {
          "passed": 5,
          "attempts": 18,
          "rate": 27.77777777777778
        }
      },
      "difficulty": {
        "Routine": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Multi-step": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        },
        "Complex": {
          "passed": 12,
          "attempts": 15,
          "rate": 80.0
        },
        "Expert": {
          "passed": 11,
          "attempts": 15,
          "rate": 73.33333333333333
        },
        "Frontier": {
          "passed": 9,
          "attempts": 15,
          "rate": 60.0
        },
        "Stress": {
          "passed": 6,
          "attempts": 15,
          "rate": 40.0
        }
      },
      "judgment": {
        "correct": 12,
        "attempts": 12,
        "correct_rate": 100.0,
        "false_stops": 0,
        "solvable_attempts": 90,
        "false_stop_rate": 0.0,
        "median_correct_stop_s": 5.5258954999998675
      },
      "predictions": {
        "brier": {
          "metric": 0.06007336541666666,
          "baseline_matched": 0.06028281498748413,
          "baseline_full": 0.06768977925381477,
          "valid_batches": 2,
          "attempted_batches": 12,
          "expected_batches": 12
        },
        "price": {
          "metric": 0.4382900567510638,
          "baseline_matched": 0.39071435980561897,
          "baseline_full": 0.3567919219293358,
          "valid_batches": 9,
          "attempted_batches": 12,
          "expected_batches": 12
        }
      },
      "strict_delivery_passed": 61,
      "parser_changed_delivery": 0,
      "parser_delivery_passed": 61,
      "semantic_changed_delivery": 0,
      "infrastructure_interruptions": 0
    }
  ],
  "default_featured_model": "openai/gpt-6-sol",
  "method_url": "/benchmarks/bid-bench-1.0-method.md",
  "candidate_coverage": {
    "actual_bidders_in_candidates": 18,
    "all_actual_bidders": 61,
    "fraction": 0.29508196721311475
  },
  "provenance": {
    "difficulty_extension": "A sixth Stress tier was frozen after a ceiling was observed on completed earlier delivery cases; all configurations receive it uniformly. This is an exploratory extension, not an untouched confirmatory set.",
    "roster_exclusion": "OpenAI Fast variants were removed at the user\u2019s request. Previously executed measurements are retained internally and excluded uniformly; 30 unstarted requests were cancelled.",
    "review_type": "Primary-agent and automated validation; no independent human review",
    "infrastructure_interruptions": 12,
    "interruption_ledger_sha256": "3317d8e538669c1b188a116574f01b8f092fb8015575da0efaebf87a2dbc2eb8",
    "interruption_treatment": "Twelve started delivery attempts lost their final receipt when the test host process stopped. Retained as unsuccessful scheduled attempts with unknown complete cost/tokens/duration; never resent or characterized as reasoning failures.",
    "graded_record_index_sha256": "591d915afd2f256776b0199c1b70562371d8f58ff666fd2614e9a66049035cab",
    "input_batches": [
      {
        "dataset_sha256": "2b52139519efb0f7d3f1afd1044c4e1b5754bc31941509c785fe0ad8d1d91d12",
        "runner_sha256": "f7290716fccfc0be0eb984fead649ceed60cd226074dcbab1f73595e49dea996",
        "scheduled_records": 7680,
        "executed_attempts": 6400,
        "cancelled_jobs": 1280
      },
      {
        "dataset_sha256": "ac53aa8b1ceb8d195427190f8aec4da8f48b8068e0c26ab3611043ef378eb555",
        "runner_sha256": "f7290716fccfc0be0eb984fead649ceed60cd226074dcbab1f73595e49dea996",
        "scheduled_records": 2160,
        "executed_attempts": 2160,
        "cancelled_jobs": 0
      },
      {
        "dataset_sha256": "7bed405d7aac3487775378dd8ed40f7d59ea392b2dfffea520a579694005d476",
        "runner_sha256": "f7290716fccfc0be0eb984fead649ceed60cd226074dcbab1f73595e49dea996",
        "scheduled_records": 1920,
        "executed_attempts": 1920,
        "cancelled_jobs": 0
      },
      {
        "dataset_sha256": "eb5b431d91070c0ae4cb180c8acc20f98abf8a413aad8cae45ecdc9b20212a44",
        "runner_sha256": "f7290716fccfc0be0eb984fead649ceed60cd226074dcbab1f73595e49dea996",
        "scheduled_records": 960,
        "executed_attempts": 960,
        "cancelled_jobs": 0
      },
      {
        "dataset_sha256": "fff94bc3e780f8cefa5e10b671bd757ca97dd7e4e1f03518c691c7640f07e68f",
        "runner_sha256": "f7290716fccfc0be0eb984fead649ceed60cd226074dcbab1f73595e49dea996",
        "scheduled_records": 1200,
        "executed_attempts": 1170,
        "cancelled_jobs": 30
      }
    ],
    "grading_correction": "Uniform final-answer extraction, verified equivalent county project-list normalization, and equivalent chronology/outcome blocker normalization for unavailable future bid prices. Original strict and parser-only pass counts are retained per configuration.",
    "fixture_correction": "Four Analysis cases were rerun uniformly with explicit approval of base estimates; numerical labels unchanged. Original ambiguous cases are excluded and retained internally.",
    "source_correction": "Original forecasts including engineer-estimate pseudo-bidders are retained internally and excluded. All configurations rerun on the corrected historical cohort."
  }
}