{
  "status": "INTERNALLY_REVIEWED_FOLLOWUP_V2_EXPORT",
  "generated_at_utc": "2026-10-08T21:52:14.985695+00:00",
  "commitment_sha256": "3ca50ab109f8e5c80fb988d7687be168dbf2c3cd7fbe51f2dabb3e44818b1a45",
  "evaluation_date_utc": "2026-10-08",
  "capacity_recovery_amendment_commitment_sha256": "ccc8140ad562e5454c65c4837212304bf25b7441d4b14b350c0fdac593681f1b",
  "concurrency_operator_commitment_sha256": null,
  "concurrency_segments": [],
  "tier_completion_commitment_sha256": "f48f049e5ec4ae71736f5d1be2265d999f81997807472185075620eef94daeb2",
  "tier_completion_reconciliation_sha256": "e9d73e12185990ff211732964225fa6607b1b29a688776719f3bd97e9661f701",
  "capacity_reconciliation_sha256": "693658a02343445f7be82115164916e1a550c423cce9137f347b2c2677ce8eb2",
  "frozen_reference_source_sha256": "259336aa866dda3fc0cd3cddb933f4b5d48025d47035bcdece9b4ec90bf7ed92",
  "first_scheduled_decision_only": true,
  "final_run_requested": true,
  "totals": {
    "scheduled_decisions": 960,
    "distinct_companies": 96,
    "outcomes": {
      "correct": 505,
      "wrong": 32,
      "cannot_verify": 421,
      "other_answer": 0,
      "operational_failed": 0,
      "missing": 0,
      "truncated": 2,
      "contaminated": 0
    },
    "operational_statuses": {
      "complete": 958,
      "truncated_output": 2
    },
    "eligible_membership_decisions": 958,
    "full_visible_answers_reviewed": 959,
    "visible_answers_needing_review": 0,
    "operational_no_answer_reviews": 1,
    "generated_no_answer_events_needing_review": 0,
    "contamination_candidates": 0,
    "reviewed_failure_stages": {
      "not_applicable": 771,
      "query_identity_error": 2,
      "unclear": 186
    },
    "recorded_paid_http_attempts": 1255,
    "recorded_api_turns": 1255,
    "recorded_http_attempts_including_prior_capacity_rejections": 1564,
    "documented_uncharged_capacity_rejections": 309,
    "generated_response_attempts": 1255,
    "decisions_with_prior_capacity_rejection": 14,
    "completed_after_prior_capacity_rejection": 14,
    "execution_phases": {
      "amendment": 74,
      "original_followup": 7,
      "tier_completion": 879
    },
    "realized_web_actions": {
      "visible_including_unfinished": {
        "n": 576,
        "median": 5.0,
        "p90": 6,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "completed": {
        "n": 576,
        "median": 4.0,
        "p90": 6,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "search_actions": {
        "n": 576,
        "median": 1.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
    },
    "cost": {
      "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
      "known_usage_based_usd": 17.741330671,
      "unknown_usage_decisions": 0,
      "observed_conservative_usd": 20.098463772,
      "budget_committed_upper_usd": 20.098463772,
      "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
      "provider_invoice_observed": false
    },
    "latency_seconds_all_recorded_decisions": {
      "n": 960,
      "median": 19.542157,
      "p90": 56.249456,
      "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
    },
    "latency_seconds_complete_eligible_decisions": {
      "n": 958,
      "median": 19.491512,
      "p90": 56.335014,
      "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
    },
    "capacity_retry_backoff_seconds": {
      "n": 960,
      "median": 0.0,
      "p90": 2,
      "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
    },
    "prior_capacity_http_latency_seconds": {
      "n": 14,
      "median": 47.4434715,
      "p90": 308.388281,
      "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
    },
    "summed_http_latency_seconds": {
      "n": 960,
      "median": 19.3799025,
      "p90": 55.948387,
      "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
    }
  },
  "by_arm": [
    {
      "arm": "D",
      "scheduled_decisions": 192,
      "distinct_companies": 96,
      "outcomes": {
        "correct": 101,
        "wrong": 0,
        "cannot_verify": 91,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 192
      },
      "eligible_membership_decisions": 192,
      "full_visible_answers_reviewed": 192,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 192
      },
      "recorded_paid_http_attempts": 192,
      "recorded_api_turns": 192,
      "recorded_http_attempts_including_prior_capacity_rejections": 213,
      "documented_uncharged_capacity_rejections": 21,
      "generated_response_attempts": 192,
      "decisions_with_prior_capacity_rejection": 2,
      "completed_after_prior_capacity_rejection": 2,
      "execution_phases": {
        "amendment": 15,
        "original_followup": 2,
        "tier_completion": 175
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.307306925,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.338037617,
        "budget_committed_upper_usd": 0.338037617,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 192,
        "median": 10.394443,
        "p90": 17.333953,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 192,
        "median": 10.394443,
        "p90": 17.333953,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 192,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 2,
        "median": 14.20524,
        "p90": 14.23762,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 192,
        "median": 10.3910765,
        "p90": 17.130564,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "arm": "T",
      "scheduled_decisions": 192,
      "distinct_companies": 96,
      "outcomes": {
        "correct": 186,
        "wrong": 0,
        "cannot_verify": 6,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 192
      },
      "eligible_membership_decisions": 192,
      "full_visible_answers_reviewed": 192,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 190,
        "query_identity_error": 2
      },
      "recorded_paid_http_attempts": 487,
      "recorded_api_turns": 487,
      "recorded_http_attempts_including_prior_capacity_rejections": 514,
      "documented_uncharged_capacity_rejections": 27,
      "generated_response_attempts": 487,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 13,
        "original_followup": 4,
        "tier_completion": 175
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.539056716,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.5929624,
        "budget_committed_upper_usd": 0.5929624,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 192,
        "median": 10.845017,
        "p90": 30.924346,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 192,
        "median": 10.845017,
        "p90": 30.924346,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 192,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 192,
        "median": 10.8367275,
        "p90": 27.360292,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "arm": "W4-high",
      "scheduled_decisions": 192,
      "distinct_companies": 96,
      "outcomes": {
        "correct": 67,
        "wrong": 14,
        "cannot_verify": 111,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 192
      },
      "eligible_membership_decisions": 192,
      "full_visible_answers_reviewed": 192,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 178,
        "unclear": 14
      },
      "recorded_paid_http_attempts": 192,
      "recorded_api_turns": 192,
      "recorded_http_attempts_including_prior_capacity_rejections": 291,
      "documented_uncharged_capacity_rejections": 99,
      "generated_response_attempts": 192,
      "decisions_with_prior_capacity_rejection": 4,
      "completed_after_prior_capacity_rejection": 4,
      "execution_phases": {
        "amendment": 16,
        "tier_completion": 176
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 192,
          "median": 5.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 192,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 192,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 5.203672205,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 5.988039437,
        "budget_committed_upper_usd": 5.988039437,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 192,
        "median": 26.770736999999997,
        "p90": 65.967727,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 192,
        "median": 26.770736999999997,
        "p90": 65.967727,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 192,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 4,
        "median": 113.0199945,
        "p90": 308.388281,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 192,
        "median": 26.7660235,
        "p90": 63.955404,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "arm": "W4-low",
      "scheduled_decisions": 192,
      "distinct_companies": 96,
      "outcomes": {
        "correct": 61,
        "wrong": 8,
        "cannot_verify": 123,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 192
      },
      "eligible_membership_decisions": 192,
      "full_visible_answers_reviewed": 192,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 61,
        "unclear": 131
      },
      "recorded_paid_http_attempts": 192,
      "recorded_api_turns": 192,
      "recorded_http_attempts_including_prior_capacity_rejections": 246,
      "documented_uncharged_capacity_rejections": 54,
      "generated_response_attempts": 192,
      "decisions_with_prior_capacity_rejection": 2,
      "completed_after_prior_capacity_rejection": 2,
      "execution_phases": {
        "amendment": 16,
        "original_followup": 1,
        "tier_completion": 175
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 192,
          "median": 5.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 192,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 192,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 5.310234515,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 6.105257974,
        "budget_committed_upper_usd": 6.105257974,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 192,
        "median": 27.067721,
        "p90": 63.103608,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 192,
        "median": 27.067721,
        "p90": 63.103608,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 192,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 2,
        "median": 47.4434715,
        "p90": 54.000262,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 192,
        "median": 27.063104,
        "p90": 63.099018,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "arm": "W8-low",
      "scheduled_decisions": 192,
      "distinct_companies": 96,
      "outcomes": {
        "correct": 90,
        "wrong": 10,
        "cannot_verify": 90,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 2,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 190,
        "truncated_output": 2
      },
      "eligible_membership_decisions": 190,
      "full_visible_answers_reviewed": 191,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 1,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 150,
        "unclear": 41
      },
      "recorded_paid_http_attempts": 192,
      "recorded_api_turns": 192,
      "recorded_http_attempts_including_prior_capacity_rejections": 300,
      "documented_uncharged_capacity_rejections": 108,
      "generated_response_attempts": 192,
      "decisions_with_prior_capacity_rejection": 6,
      "completed_after_prior_capacity_rejection": 6,
      "execution_phases": {
        "amendment": 14,
        "tier_completion": 178
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 192,
          "median": 5.0,
          "p90": 8,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 192,
          "median": 5.0,
          "p90": 8,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 192,
          "median": 1.0,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 6.38106031,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 7.074166344,
        "budget_committed_upper_usd": 7.074166344,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 192,
        "median": 29.867307,
        "p90": 76.072468,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 190,
        "median": 29.764344,
        "p90": 76.072468,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 192,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 6,
        "median": 161.672065,
        "p90": 357.0459649999999,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 192,
        "median": 29.8652115,
        "p90": 76.067027,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    }
  ],
  "by_model_arm": [
    {
      "model": "gpt-6-luna",
      "arm": "D",
      "scheduled_decisions": 96,
      "distinct_companies": 96,
      "outcomes": {
        "correct": 36,
        "wrong": 0,
        "cannot_verify": 60,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 96
      },
      "eligible_membership_decisions": 96,
      "full_visible_answers_reviewed": 96,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 96
      },
      "recorded_paid_http_attempts": 96,
      "recorded_api_turns": 96,
      "recorded_http_attempts_including_prior_capacity_rejections": 102,
      "documented_uncharged_capacity_rejections": 6,
      "generated_response_attempts": 96,
      "decisions_with_prior_capacity_rejection": 2,
      "completed_after_prior_capacity_rejection": 2,
      "execution_phases": {
        "amendment": 9,
        "tier_completion": 87
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.030234275,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.033257702,
        "budget_committed_upper_usd": 0.033257702,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 96,
        "median": 5.9974205000000005,
        "p90": 14.267172,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 96,
        "median": 5.9974205000000005,
        "p90": 14.267172,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 96,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 2,
        "median": 14.20524,
        "p90": 14.23762,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 96,
        "median": 5.992949,
        "p90": 14.262019,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "T",
      "scheduled_decisions": 96,
      "distinct_companies": 96,
      "outcomes": {
        "correct": 91,
        "wrong": 0,
        "cannot_verify": 5,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 96
      },
      "eligible_membership_decisions": 96,
      "full_visible_answers_reviewed": 96,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 94,
        "query_identity_error": 2
      },
      "recorded_paid_http_attempts": 246,
      "recorded_api_turns": 246,
      "recorded_http_attempts_including_prior_capacity_rejections": 253,
      "documented_uncharged_capacity_rejections": 7,
      "generated_response_attempts": 246,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 7,
        "original_followup": 2,
        "tier_completion": 87
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.052356416,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.05759207,
        "budget_committed_upper_usd": 0.05759207,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 96,
        "median": 7.530533999999999,
        "p90": 12.915276,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 96,
        "median": 7.530533999999999,
        "p90": 12.915276,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 96,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 96,
        "median": 7.522459,
        "p90": 12.904765999999999,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-high",
      "scheduled_decisions": 96,
      "distinct_companies": 96,
      "outcomes": {
        "correct": 21,
        "wrong": 14,
        "cannot_verify": 61,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 96
      },
      "eligible_membership_decisions": 96,
      "full_visible_answers_reviewed": 96,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 82,
        "unclear": 14
      },
      "recorded_paid_http_attempts": 96,
      "recorded_api_turns": 96,
      "recorded_http_attempts_including_prior_capacity_rejections": 151,
      "documented_uncharged_capacity_rejections": 55,
      "generated_response_attempts": 96,
      "decisions_with_prior_capacity_rejection": 3,
      "completed_after_prior_capacity_rejection": 3,
      "execution_phases": {
        "amendment": 7,
        "tier_completion": 89
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 96,
          "median": 5.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 96,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 96,
          "median": 2.0,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 1.642987805,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 2.049286597,
        "budget_committed_upper_usd": 2.049286597,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 96,
        "median": 21.793962999999998,
        "p90": 30.44864,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 96,
        "median": 21.793962999999998,
        "p90": 30.44864,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 96,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 3,
        "median": 212.670827,
        "p90": 308.388281,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 96,
        "median": 21.7890345,
        "p90": 30.443101,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-low",
      "scheduled_decisions": 96,
      "distinct_companies": 96,
      "outcomes": {
        "correct": 22,
        "wrong": 8,
        "cannot_verify": 66,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 96
      },
      "eligible_membership_decisions": 96,
      "full_visible_answers_reviewed": 96,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 22,
        "unclear": 74
      },
      "recorded_paid_http_attempts": 96,
      "recorded_api_turns": 96,
      "recorded_http_attempts_including_prior_capacity_rejections": 115,
      "documented_uncharged_capacity_rejections": 19,
      "generated_response_attempts": 96,
      "decisions_with_prior_capacity_rejection": 2,
      "completed_after_prior_capacity_rejection": 2,
      "execution_phases": {
        "amendment": 8,
        "tier_completion": 88
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 96,
          "median": 5.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 96,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 96,
          "median": 2.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 1.654585915,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 2.040044514,
        "budget_committed_upper_usd": 2.040044514,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 96,
        "median": 19.686318,
        "p90": 34.538759,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 96,
        "median": 19.686318,
        "p90": 34.538759,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 96,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 2,
        "median": 47.4434715,
        "p90": 54.000262,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 96,
        "median": 19.6811875,
        "p90": 34.533149,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W8-low",
      "scheduled_decisions": 96,
      "distinct_companies": 96,
      "outcomes": {
        "correct": 35,
        "wrong": 10,
        "cannot_verify": 49,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 2,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 94,
        "truncated_output": 2
      },
      "eligible_membership_decisions": 94,
      "full_visible_answers_reviewed": 95,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 1,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 85,
        "unclear": 10
      },
      "recorded_paid_http_attempts": 96,
      "recorded_api_turns": 96,
      "recorded_http_attempts_including_prior_capacity_rejections": 155,
      "documented_uncharged_capacity_rejections": 59,
      "generated_response_attempts": 96,
      "decisions_with_prior_capacity_rejection": 5,
      "completed_after_prior_capacity_rejection": 5,
      "execution_phases": {
        "amendment": 6,
        "tier_completion": 90
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 96,
          "median": 5.0,
          "p90": 9,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 96,
          "median": 5.0,
          "p90": 8,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 96,
          "median": 2.0,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 2.21055371,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 2.486609084,
        "budget_committed_upper_usd": 2.486609084,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 96,
        "median": 21.318904500000002,
        "p90": 41.878121,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 94,
        "median": 21.182993,
        "p90": 39.628916,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 96,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 5,
        "median": 267.94633600000003,
        "p90": 357.0459649999999,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 96,
        "median": 21.3153805,
        "p90": 41.870892,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "D",
      "scheduled_decisions": 96,
      "distinct_companies": 96,
      "outcomes": {
        "correct": 65,
        "wrong": 0,
        "cannot_verify": 31,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 96
      },
      "eligible_membership_decisions": 96,
      "full_visible_answers_reviewed": 96,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 96
      },
      "recorded_paid_http_attempts": 96,
      "recorded_api_turns": 96,
      "recorded_http_attempts_including_prior_capacity_rejections": 111,
      "documented_uncharged_capacity_rejections": 15,
      "generated_response_attempts": 96,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 6,
        "original_followup": 2,
        "tier_completion": 88
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.27707265,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.304779915,
        "budget_committed_upper_usd": 0.304779915,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 96,
        "median": 13.361460000000001,
        "p90": 18.007425,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 96,
        "median": 13.361460000000001,
        "p90": 18.007425,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 96,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 96,
        "median": 13.284182999999999,
        "p90": 17.716105,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "T",
      "scheduled_decisions": 96,
      "distinct_companies": 96,
      "outcomes": {
        "correct": 95,
        "wrong": 0,
        "cannot_verify": 1,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 96
      },
      "eligible_membership_decisions": 96,
      "full_visible_answers_reviewed": 96,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 96
      },
      "recorded_paid_http_attempts": 241,
      "recorded_api_turns": 241,
      "recorded_http_attempts_including_prior_capacity_rejections": 261,
      "documented_uncharged_capacity_rejections": 20,
      "generated_response_attempts": 241,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 6,
        "original_followup": 2,
        "tier_completion": 88
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.4867003,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.53537033,
        "budget_committed_upper_usd": 0.53537033,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 96,
        "median": 17.2316485,
        "p90": 31.872571,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 96,
        "median": 17.2316485,
        "p90": 31.872571,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 96,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 96,
        "median": 17.2239065,
        "p90": 29.481018000000002,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-high",
      "scheduled_decisions": 96,
      "distinct_companies": 96,
      "outcomes": {
        "correct": 46,
        "wrong": 0,
        "cannot_verify": 50,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 96
      },
      "eligible_membership_decisions": 96,
      "full_visible_answers_reviewed": 96,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 96
      },
      "recorded_paid_http_attempts": 96,
      "recorded_api_turns": 96,
      "recorded_http_attempts_including_prior_capacity_rejections": 140,
      "documented_uncharged_capacity_rejections": 44,
      "generated_response_attempts": 96,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 9,
        "tier_completion": 87
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 96,
          "median": 5.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 96,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 96,
          "median": 1.0,
          "p90": 1,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 3.5606844,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 3.93875284,
        "budget_committed_upper_usd": 3.93875284,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 96,
        "median": 40.9654135,
        "p90": 80.362308,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 96,
        "median": 40.9654135,
        "p90": 80.362308,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 96,
        "median": 0.0,
        "p90": 6,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 11.049299,
        "p90": 11.049299,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 96,
        "median": 40.8222735,
        "p90": 73.810778,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-low",
      "scheduled_decisions": 96,
      "distinct_companies": 96,
      "outcomes": {
        "correct": 39,
        "wrong": 0,
        "cannot_verify": 57,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 96
      },
      "eligible_membership_decisions": 96,
      "full_visible_answers_reviewed": 96,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 39,
        "unclear": 57
      },
      "recorded_paid_http_attempts": 96,
      "recorded_api_turns": 96,
      "recorded_http_attempts_including_prior_capacity_rejections": 131,
      "documented_uncharged_capacity_rejections": 35,
      "generated_response_attempts": 96,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 8,
        "original_followup": 1,
        "tier_completion": 87
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 96,
          "median": 5.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 96,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 96,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 3.6556486,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 4.06521346,
        "budget_committed_upper_usd": 4.06521346,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 96,
        "median": 42.6963,
        "p90": 76.2564,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 96,
        "median": 42.6963,
        "p90": 76.2564,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 96,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 96,
        "median": 42.339066,
        "p90": 73.118786,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W8-low",
      "scheduled_decisions": 96,
      "distinct_companies": 96,
      "outcomes": {
        "correct": 55,
        "wrong": 0,
        "cannot_verify": 41,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 96
      },
      "eligible_membership_decisions": 96,
      "full_visible_answers_reviewed": 96,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 65,
        "unclear": 31
      },
      "recorded_paid_http_attempts": 96,
      "recorded_api_turns": 96,
      "recorded_http_attempts_including_prior_capacity_rejections": 145,
      "documented_uncharged_capacity_rejections": 49,
      "generated_response_attempts": 96,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 8,
        "tier_completion": 88
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 96,
          "median": 5.0,
          "p90": 6,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 96,
          "median": 5.0,
          "p90": 6,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 96,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 4.1705066,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 4.58755726,
        "budget_committed_upper_usd": 4.58755726,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 96,
        "median": 43.8265205,
        "p90": 100.087125,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 96,
        "median": 43.8265205,
        "p90": 100.087125,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 96,
        "median": 0.0,
        "p90": 6,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 11.064242,
        "p90": 11.064242,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 96,
        "median": 43.8215875,
        "p90": 94.058713,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    }
  ],
  "by_model_arm_cohort": [
    {
      "model": "gpt-6-luna",
      "arm": "D",
      "cohort": "current_control",
      "scheduled_decisions": 32,
      "distinct_companies": 32,
      "outcomes": {
        "correct": 32,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 32
      },
      "eligible_membership_decisions": 32,
      "full_visible_answers_reviewed": 32,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 32
      },
      "recorded_paid_http_attempts": 32,
      "recorded_api_turns": 32,
      "recorded_http_attempts_including_prior_capacity_rejections": 34,
      "documented_uncharged_capacity_rejections": 2,
      "generated_response_attempts": 32,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 5,
        "tier_completion": 27
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.00833715,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.009170865,
        "budget_committed_upper_usd": 0.009170865,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 32,
        "median": 4.711835499999999,
        "p90": 15.382581,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 32,
        "median": 4.711835499999999,
        "p90": 15.382581,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 32,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 14.23762,
        "p90": 14.23762,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 32,
        "median": 4.7080044999999995,
        "p90": 15.378348,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "D",
      "cohort": "removed",
      "scheduled_decisions": 64,
      "distinct_companies": 64,
      "outcomes": {
        "correct": 4,
        "wrong": 0,
        "cannot_verify": 60,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 64
      },
      "eligible_membership_decisions": 64,
      "full_visible_answers_reviewed": 64,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 64
      },
      "recorded_paid_http_attempts": 64,
      "recorded_api_turns": 64,
      "recorded_http_attempts_including_prior_capacity_rejections": 68,
      "documented_uncharged_capacity_rejections": 4,
      "generated_response_attempts": 64,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 4,
        "tier_completion": 60
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.021897125,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.024086837,
        "budget_committed_upper_usd": 0.024086837,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 64,
        "median": 6.5219465,
        "p90": 12.039162,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 64,
        "median": 6.5219465,
        "p90": 12.039162,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 64,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 14.17286,
        "p90": 14.17286,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 64,
        "median": 6.518020999999999,
        "p90": 12.036095,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "T",
      "cohort": "current_control",
      "scheduled_decisions": 32,
      "distinct_companies": 32,
      "outcomes": {
        "correct": 32,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 32
      },
      "eligible_membership_decisions": 32,
      "full_visible_answers_reviewed": 32,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 32
      },
      "recorded_paid_http_attempts": 79,
      "recorded_api_turns": 79,
      "recorded_http_attempts_including_prior_capacity_rejections": 81,
      "documented_uncharged_capacity_rejections": 2,
      "generated_response_attempts": 79,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 4,
        "original_followup": 1,
        "tier_completion": 27
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.014322038,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.015754247,
        "budget_committed_upper_usd": 0.015754247,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 32,
        "median": 6.30624,
        "p90": 22.623293,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 32,
        "median": 6.30624,
        "p90": 22.623293,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 32,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 32,
        "median": 6.2961525,
        "p90": 22.619500000000002,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "T",
      "cohort": "removed",
      "scheduled_decisions": 64,
      "distinct_companies": 64,
      "outcomes": {
        "correct": 59,
        "wrong": 0,
        "cannot_verify": 5,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 64
      },
      "eligible_membership_decisions": 64,
      "full_visible_answers_reviewed": 64,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 62,
        "query_identity_error": 2
      },
      "recorded_paid_http_attempts": 167,
      "recorded_api_turns": 167,
      "recorded_http_attempts_including_prior_capacity_rejections": 172,
      "documented_uncharged_capacity_rejections": 5,
      "generated_response_attempts": 167,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 3,
        "original_followup": 1,
        "tier_completion": 60
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.038034378,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.041837823,
        "budget_committed_upper_usd": 0.041837823,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 64,
        "median": 8.454405,
        "p90": 11.461684,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 64,
        "median": 8.454405,
        "p90": 11.461684,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 64,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 64,
        "median": 8.446695,
        "p90": 11.452494000000002,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-high",
      "cohort": "current_control",
      "scheduled_decisions": 32,
      "distinct_companies": 32,
      "outcomes": {
        "correct": 17,
        "wrong": 0,
        "cannot_verify": 15,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 32
      },
      "eligible_membership_decisions": 32,
      "full_visible_answers_reviewed": 32,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 32
      },
      "recorded_paid_http_attempts": 32,
      "recorded_api_turns": 32,
      "recorded_http_attempts_including_prior_capacity_rejections": 53,
      "documented_uncharged_capacity_rejections": 21,
      "generated_response_attempts": 32,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 5,
        "tier_completion": 27
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 32,
          "median": 3.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 32,
          "median": 3.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 32,
          "median": 2.0,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.570008715,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.671009591,
        "budget_committed_upper_usd": 0.671009591,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 32,
        "median": 17.243733499999998,
        "p90": 136.902643,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 32,
        "median": 17.243733499999998,
        "p90": 136.902643,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 32,
        "median": 0.0,
        "p90": 28,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 13.369162,
        "p90": 13.369162,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 32,
        "median": 17.239549500000003,
        "p90": 100.949592,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-high",
      "cohort": "removed",
      "scheduled_decisions": 64,
      "distinct_companies": 64,
      "outcomes": {
        "correct": 4,
        "wrong": 14,
        "cannot_verify": 46,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 64
      },
      "eligible_membership_decisions": 64,
      "full_visible_answers_reviewed": 64,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 50,
        "unclear": 14
      },
      "recorded_paid_http_attempts": 64,
      "recorded_api_turns": 64,
      "recorded_http_attempts_including_prior_capacity_rejections": 98,
      "documented_uncharged_capacity_rejections": 34,
      "generated_response_attempts": 64,
      "decisions_with_prior_capacity_rejection": 2,
      "completed_after_prior_capacity_rejection": 2,
      "execution_phases": {
        "amendment": 2,
        "tier_completion": 62
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 64,
          "median": 5.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 64,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 64,
          "median": 2.0,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 1.07297909,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 1.378277006,
        "budget_committed_upper_usd": 1.378277006,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 64,
        "median": 22.6428565,
        "p90": 29.38669,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 64,
        "median": 22.6428565,
        "p90": 29.38669,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 64,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 2,
        "median": 260.529554,
        "p90": 308.388281,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 64,
        "median": 22.639383000000002,
        "p90": 29.38171,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-low",
      "cohort": "current_control",
      "scheduled_decisions": 32,
      "distinct_companies": 32,
      "outcomes": {
        "correct": 18,
        "wrong": 0,
        "cannot_verify": 14,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 32
      },
      "eligible_membership_decisions": 32,
      "full_visible_answers_reviewed": 32,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 18,
        "unclear": 14
      },
      "recorded_paid_http_attempts": 32,
      "recorded_api_turns": 32,
      "recorded_http_attempts_including_prior_capacity_rejections": 38,
      "documented_uncharged_capacity_rejections": 6,
      "generated_response_attempts": 32,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 4,
        "tier_completion": 28
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 32,
          "median": 3.5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 32,
          "median": 3.5,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 32,
          "median": 2.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.520029303,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.660032235,
        "budget_committed_upper_usd": 0.660032235,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 32,
        "median": 15.9089075,
        "p90": 37.776489,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 32,
        "median": 15.9089075,
        "p90": 37.776489,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 32,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 54.000262,
        "p90": 54.000262,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 32,
        "median": 15.90429,
        "p90": 37.774957,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-low",
      "cohort": "removed",
      "scheduled_decisions": 64,
      "distinct_companies": 64,
      "outcomes": {
        "correct": 4,
        "wrong": 8,
        "cannot_verify": 52,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 64
      },
      "eligible_membership_decisions": 64,
      "full_visible_answers_reviewed": 64,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 4,
        "unclear": 60
      },
      "recorded_paid_http_attempts": 64,
      "recorded_api_turns": 64,
      "recorded_http_attempts_including_prior_capacity_rejections": 77,
      "documented_uncharged_capacity_rejections": 13,
      "generated_response_attempts": 64,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 4,
        "tier_completion": 60
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 64,
          "median": 5.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 64,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 64,
          "median": 2.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 1.134556612,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 1.380012279,
        "budget_committed_upper_usd": 1.380012279,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 64,
        "median": 21.566922499999997,
        "p90": 30.194231,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 64,
        "median": 21.566922499999997,
        "p90": 30.194231,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 64,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 40.886681,
        "p90": 40.886681,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 64,
        "median": 21.562528999999998,
        "p90": 30.188456,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W8-low",
      "cohort": "current_control",
      "scheduled_decisions": 32,
      "distinct_companies": 32,
      "outcomes": {
        "correct": 24,
        "wrong": 0,
        "cannot_verify": 8,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 32
      },
      "eligible_membership_decisions": 32,
      "full_visible_answers_reviewed": 32,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 32
      },
      "recorded_paid_http_attempts": 32,
      "recorded_api_turns": 32,
      "recorded_http_attempts_including_prior_capacity_rejections": 53,
      "documented_uncharged_capacity_rejections": 21,
      "generated_response_attempts": 32,
      "decisions_with_prior_capacity_rejection": 2,
      "completed_after_prior_capacity_rejection": 2,
      "execution_phases": {
        "amendment": 4,
        "tier_completion": 28
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 32,
          "median": 2.5,
          "p90": 6,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 32,
          "median": 2.5,
          "p90": 6,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 32,
          "median": 2.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.62650067,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.689150738,
        "budget_committed_upper_usd": 0.689150738,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 32,
        "median": 15.3884765,
        "p90": 38.955046,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 32,
        "median": 15.3884765,
        "p90": 38.955046,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 32,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 2,
        "median": 169.598218,
        "p90": 283.79864200000003,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 32,
        "median": 15.383578499999999,
        "p90": 38.947875,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W8-low",
      "cohort": "removed",
      "scheduled_decisions": 64,
      "distinct_companies": 64,
      "outcomes": {
        "correct": 11,
        "wrong": 10,
        "cannot_verify": 41,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 2,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 62,
        "truncated_output": 2
      },
      "eligible_membership_decisions": 62,
      "full_visible_answers_reviewed": 63,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 1,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 53,
        "unclear": 10
      },
      "recorded_paid_http_attempts": 64,
      "recorded_api_turns": 64,
      "recorded_http_attempts_including_prior_capacity_rejections": 102,
      "documented_uncharged_capacity_rejections": 38,
      "generated_response_attempts": 64,
      "decisions_with_prior_capacity_rejection": 3,
      "completed_after_prior_capacity_rejection": 3,
      "execution_phases": {
        "amendment": 2,
        "tier_completion": 62
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 64,
          "median": 6.0,
          "p90": 9,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 64,
          "median": 6.0,
          "p90": 8,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 64,
          "median": 2.0,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 1.58405304,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 1.797458346,
        "budget_committed_upper_usd": 1.797458346,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 64,
        "median": 25.8645295,
        "p90": 41.878121,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 62,
        "median": 25.097680500000003,
        "p90": 39.628916,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 64,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 3,
        "median": 267.94633600000003,
        "p90": 357.0459649999999,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 64,
        "median": 25.861004,
        "p90": 41.870892,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "D",
      "cohort": "current_control",
      "scheduled_decisions": 32,
      "distinct_companies": 32,
      "outcomes": {
        "correct": 32,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 32
      },
      "eligible_membership_decisions": 32,
      "full_visible_answers_reviewed": 32,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 32
      },
      "recorded_paid_http_attempts": 32,
      "recorded_api_turns": 32,
      "recorded_http_attempts_including_prior_capacity_rejections": 37,
      "documented_uncharged_capacity_rejections": 5,
      "generated_response_attempts": 32,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 3,
        "original_followup": 1,
        "tier_completion": 28
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.0771927,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.08491197,
        "budget_committed_upper_usd": 0.08491197,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 32,
        "median": 12.0626265,
        "p90": 15.516477,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 32,
        "median": 12.0626265,
        "p90": 15.516477,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 32,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 32,
        "median": 11.9866075,
        "p90": 13.503425,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "D",
      "cohort": "removed",
      "scheduled_decisions": 64,
      "distinct_companies": 64,
      "outcomes": {
        "correct": 33,
        "wrong": 0,
        "cannot_verify": 31,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 64
      },
      "eligible_membership_decisions": 64,
      "full_visible_answers_reviewed": 64,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 64
      },
      "recorded_paid_http_attempts": 64,
      "recorded_api_turns": 64,
      "recorded_http_attempts_including_prior_capacity_rejections": 74,
      "documented_uncharged_capacity_rejections": 10,
      "generated_response_attempts": 64,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 3,
        "original_followup": 1,
        "tier_completion": 60
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.19987995,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.219867945,
        "budget_committed_upper_usd": 0.219867945,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 64,
        "median": 14.726701,
        "p90": 18.509505,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 64,
        "median": 14.726701,
        "p90": 18.509505,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 64,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 64,
        "median": 14.5193185,
        "p90": 17.865896,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "T",
      "cohort": "current_control",
      "scheduled_decisions": 32,
      "distinct_companies": 32,
      "outcomes": {
        "correct": 32,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 32
      },
      "eligible_membership_decisions": 32,
      "full_visible_answers_reviewed": 32,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 32
      },
      "recorded_paid_http_attempts": 64,
      "recorded_api_turns": 64,
      "recorded_http_attempts_including_prior_capacity_rejections": 69,
      "documented_uncharged_capacity_rejections": 5,
      "generated_response_attempts": 64,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 3,
        "original_followup": 1,
        "tier_completion": 28
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.11492205,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.126414255,
        "budget_committed_upper_usd": 0.126414255,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 32,
        "median": 11.910858999999999,
        "p90": 22.199687,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 32,
        "median": 11.910858999999999,
        "p90": 22.199687,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 32,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 32,
        "median": 11.9032,
        "p90": 22.193234,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "T",
      "cohort": "removed",
      "scheduled_decisions": 64,
      "distinct_companies": 64,
      "outcomes": {
        "correct": 63,
        "wrong": 0,
        "cannot_verify": 1,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 64
      },
      "eligible_membership_decisions": 64,
      "full_visible_answers_reviewed": 64,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 64
      },
      "recorded_paid_http_attempts": 177,
      "recorded_api_turns": 177,
      "recorded_http_attempts_including_prior_capacity_rejections": 192,
      "documented_uncharged_capacity_rejections": 15,
      "generated_response_attempts": 177,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 3,
        "original_followup": 1,
        "tier_completion": 60
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.37177825,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.408956075,
        "budget_committed_upper_usd": 0.408956075,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 64,
        "median": 19.1598215,
        "p90": 33.200919,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 64,
        "median": 19.1598215,
        "p90": 33.200919,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 64,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 64,
        "median": 18.912311,
        "p90": 31.814814,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-high",
      "cohort": "current_control",
      "scheduled_decisions": 32,
      "distinct_companies": 32,
      "outcomes": {
        "correct": 28,
        "wrong": 0,
        "cannot_verify": 4,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 32
      },
      "eligible_membership_decisions": 32,
      "full_visible_answers_reviewed": 32,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 32
      },
      "recorded_paid_http_attempts": 32,
      "recorded_api_turns": 32,
      "recorded_http_attempts_including_prior_capacity_rejections": 41,
      "documented_uncharged_capacity_rejections": 9,
      "generated_response_attempts": 32,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 5,
        "tier_completion": 27
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 32,
          "median": 2.5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 32,
          "median": 2.5,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 32,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.8486674,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.93353414,
        "budget_committed_upper_usd": 0.93353414,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 32,
        "median": 30.18275,
        "p90": 47.042037,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 32,
        "median": 30.18275,
        "p90": 47.042037,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 32,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 11.049299,
        "p90": 11.049299,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 32,
        "median": 30.178593,
        "p90": 45.029414,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-high",
      "cohort": "removed",
      "scheduled_decisions": 64,
      "distinct_companies": 64,
      "outcomes": {
        "correct": 18,
        "wrong": 0,
        "cannot_verify": 46,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 64
      },
      "eligible_membership_decisions": 64,
      "full_visible_answers_reviewed": 64,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 64
      },
      "recorded_paid_http_attempts": 64,
      "recorded_api_turns": 64,
      "recorded_http_attempts_including_prior_capacity_rejections": 99,
      "documented_uncharged_capacity_rejections": 35,
      "generated_response_attempts": 64,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 4,
        "tier_completion": 60
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 64,
          "median": 5.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 64,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 64,
          "median": 1.0,
          "p90": 1,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 2.712017,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 3.0052187,
        "budget_committed_upper_usd": 3.0052187,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 64,
        "median": 48.658565,
        "p90": 87.914627,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 64,
        "median": 48.658565,
        "p90": 87.914627,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 64,
        "median": 0.0,
        "p90": 6,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 64,
        "median": 48.654151999999996,
        "p90": 79.807165,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-low",
      "cohort": "current_control",
      "scheduled_decisions": 32,
      "distinct_companies": 32,
      "outcomes": {
        "correct": 23,
        "wrong": 0,
        "cannot_verify": 9,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 32
      },
      "eligible_membership_decisions": 32,
      "full_visible_answers_reviewed": 32,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 23,
        "unclear": 9
      },
      "recorded_paid_http_attempts": 32,
      "recorded_api_turns": 32,
      "recorded_http_attempts_including_prior_capacity_rejections": 42,
      "documented_uncharged_capacity_rejections": 10,
      "generated_response_attempts": 32,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 4,
        "original_followup": 1,
        "tier_completion": 27
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 32,
          "median": 3.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 32,
          "median": 3.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 32,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.8728554,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.96014094,
        "budget_committed_upper_usd": 0.96014094,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 32,
        "median": 32.1392565,
        "p90": 57.944792,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 32,
        "median": 32.1392565,
        "p90": 57.944792,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 32,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 32,
        "median": 32.136202,
        "p90": 57.939977,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-low",
      "cohort": "removed",
      "scheduled_decisions": 64,
      "distinct_companies": 64,
      "outcomes": {
        "correct": 16,
        "wrong": 0,
        "cannot_verify": 48,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 64
      },
      "eligible_membership_decisions": 64,
      "full_visible_answers_reviewed": 64,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 16,
        "unclear": 48
      },
      "recorded_paid_http_attempts": 64,
      "recorded_api_turns": 64,
      "recorded_http_attempts_including_prior_capacity_rejections": 89,
      "documented_uncharged_capacity_rejections": 25,
      "generated_response_attempts": 64,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 4,
        "tier_completion": 60
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 64,
          "median": 5.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 64,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 64,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 2.7827932,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 3.10507252,
        "budget_committed_upper_usd": 3.10507252,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 64,
        "median": 46.160909000000004,
        "p90": 79.141097,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 64,
        "median": 46.160909000000004,
        "p90": 79.141097,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 64,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 64,
        "median": 46.156666,
        "p90": 73.302651,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W8-low",
      "cohort": "current_control",
      "scheduled_decisions": 32,
      "distinct_companies": 32,
      "outcomes": {
        "correct": 28,
        "wrong": 0,
        "cannot_verify": 4,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 32
      },
      "eligible_membership_decisions": 32,
      "full_visible_answers_reviewed": 32,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 28,
        "unclear": 4
      },
      "recorded_paid_http_attempts": 32,
      "recorded_api_turns": 32,
      "recorded_http_attempts_including_prior_capacity_rejections": 40,
      "documented_uncharged_capacity_rejections": 8,
      "generated_response_attempts": 32,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 4,
        "tier_completion": 28
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 32,
          "median": 2.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 32,
          "median": 2.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 32,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.8758986,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.96348846,
        "budget_committed_upper_usd": 0.96348846,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 32,
        "median": 29.653719000000002,
        "p90": 62.981264,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 32,
        "median": 29.653719000000002,
        "p90": 62.981264,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 32,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 11.064242,
        "p90": 11.064242,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 32,
        "median": 29.6506755,
        "p90": 60.971222,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W8-low",
      "cohort": "removed",
      "scheduled_decisions": 64,
      "distinct_companies": 64,
      "outcomes": {
        "correct": 27,
        "wrong": 0,
        "cannot_verify": 37,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 64
      },
      "eligible_membership_decisions": 64,
      "full_visible_answers_reviewed": 64,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 37,
        "unclear": 27
      },
      "recorded_paid_http_attempts": 64,
      "recorded_api_turns": 64,
      "recorded_http_attempts_including_prior_capacity_rejections": 105,
      "documented_uncharged_capacity_rejections": 41,
      "generated_response_attempts": 64,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 4,
        "tier_completion": 60
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 64,
          "median": 5.0,
          "p90": 7,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 64,
          "median": 5.0,
          "p90": 7,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 64,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 3.294608,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 3.6240688,
        "budget_committed_upper_usd": 3.6240688,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 64,
        "median": 49.8555835,
        "p90": 118.842583,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 64,
        "median": 49.8555835,
        "p90": 118.842583,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 64,
        "median": 0.0,
        "p90": 14,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 64,
        "median": 49.850705000000005,
        "p90": 108.55159499999999,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    }
  ],
  "by_model_arm_stratum": [
    {
      "model": "gpt-6-luna",
      "arm": "D",
      "stratum": "listed-control-IRAQ2",
      "scheduled_decisions": 8,
      "distinct_companies": 8,
      "outcomes": {
        "correct": 8,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 8
      },
      "eligible_membership_decisions": 8,
      "full_visible_answers_reviewed": 8,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 8
      },
      "recorded_paid_http_attempts": 8,
      "recorded_api_turns": 8,
      "recorded_http_attempts_including_prior_capacity_rejections": 10,
      "documented_uncharged_capacity_rejections": 2,
      "generated_response_attempts": 8,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 2,
        "tier_completion": 6
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.0018552,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.00204072,
        "budget_committed_upper_usd": 0.00204072,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 8,
        "median": 4.904621499999999,
        "p90": 26.928276,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 8,
        "median": 4.904621499999999,
        "p90": 26.928276,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 8,
        "median": 0.0,
        "p90": 4,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 14.23762,
        "p90": 14.23762,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 8,
        "median": 4.9004005,
        "p90": 22.920596,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "D",
      "stratum": "listed-control-SDNT",
      "scheduled_decisions": 20,
      "distinct_companies": 20,
      "outcomes": {
        "correct": 20,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 20
      },
      "eligible_membership_decisions": 20,
      "full_visible_answers_reviewed": 20,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 20
      },
      "recorded_paid_http_attempts": 20,
      "recorded_api_turns": 20,
      "recorded_http_attempts_including_prior_capacity_rejections": 20,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 20,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 19
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.0056687,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.00623557,
        "budget_committed_upper_usd": 0.00623557,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 20,
        "median": 4.694851,
        "p90": 5.445513,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 20,
        "median": 4.694851,
        "p90": 5.445513,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 20,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 20,
        "median": 4.690156,
        "p90": 5.442128,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "D",
      "stratum": "listed-control-SDNTK",
      "scheduled_decisions": 4,
      "distinct_companies": 4,
      "outcomes": {
        "correct": 4,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 4
      },
      "eligible_membership_decisions": 4,
      "full_visible_answers_reviewed": 4,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 4
      },
      "recorded_paid_http_attempts": 4,
      "recorded_api_turns": 4,
      "recorded_http_attempts_including_prior_capacity_rejections": 4,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 4,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 2,
        "tier_completion": 2
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.00081325,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.000894575,
        "budget_committed_upper_usd": 0.000894575,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 4,
        "median": 9.8012515,
        "p90": 18.69913,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 4,
        "median": 9.8012515,
        "p90": 18.69913,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 4,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 4,
        "median": 9.7980385,
        "p90": 18.697637,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "D",
      "stratum": "removed-20260528",
      "scheduled_decisions": 9,
      "distinct_companies": 9,
      "outcomes": {
        "correct": 0,
        "wrong": 0,
        "cannot_verify": 9,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 9
      },
      "eligible_membership_decisions": 9,
      "full_visible_answers_reviewed": 9,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 9
      },
      "recorded_paid_http_attempts": 9,
      "recorded_api_turns": 9,
      "recorded_http_attempts_including_prior_capacity_rejections": 12,
      "documented_uncharged_capacity_rejections": 3,
      "generated_response_attempts": 9,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 2,
        "tier_completion": 7
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.00249845,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.002748295,
        "budget_committed_upper_usd": 0.002748295,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 9,
        "median": 6.4837,
        "p90": 45.676921,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 9,
        "median": 6.4837,
        "p90": 45.676921,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 9,
        "median": 0,
        "p90": 12,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 14.17286,
        "p90": 14.17286,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 9,
        "median": 6.479979,
        "p90": 33.664190000000005,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "D",
      "stratum": "removed-20260727",
      "scheduled_decisions": 27,
      "distinct_companies": 27,
      "outcomes": {
        "correct": 0,
        "wrong": 0,
        "cannot_verify": 27,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 27
      },
      "eligible_membership_decisions": 27,
      "full_visible_answers_reviewed": 27,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 27
      },
      "recorded_paid_http_attempts": 27,
      "recorded_api_turns": 27,
      "recorded_http_attempts_including_prior_capacity_rejections": 28,
      "documented_uncharged_capacity_rejections": 1,
      "generated_response_attempts": 27,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 26
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.00825405,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.009079455,
        "budget_committed_upper_usd": 0.009079455,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 27,
        "median": 6.028579,
        "p90": 7.683599,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 27,
        "median": 6.028579,
        "p90": 7.683599,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 27,
        "median": 0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 27,
        "median": 6.023596,
        "p90": 7.679706,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "D",
      "stratum": "removed-20261005",
      "scheduled_decisions": 28,
      "distinct_companies": 28,
      "outcomes": {
        "correct": 4,
        "wrong": 0,
        "cannot_verify": 24,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 28
      },
      "eligible_membership_decisions": 28,
      "full_visible_answers_reviewed": 28,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 28
      },
      "recorded_paid_http_attempts": 28,
      "recorded_api_turns": 28,
      "recorded_http_attempts_including_prior_capacity_rejections": 28,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 28,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 27
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.011144625,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.012259087,
        "budget_committed_upper_usd": 0.012259087,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 28,
        "median": 7.3199345000000005,
        "p90": 13.399748,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 28,
        "median": 7.3199345000000005,
        "p90": 13.399748,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 28,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 28,
        "median": 7.3155475,
        "p90": 13.396023,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "T",
      "stratum": "listed-control-IRAQ2",
      "scheduled_decisions": 8,
      "distinct_companies": 8,
      "outcomes": {
        "correct": 8,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 8
      },
      "eligible_membership_decisions": 8,
      "full_visible_answers_reviewed": 8,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 8
      },
      "recorded_paid_http_attempts": 16,
      "recorded_api_turns": 16,
      "recorded_http_attempts_including_prior_capacity_rejections": 16,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 16,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "original_followup": 1,
        "tier_completion": 6
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.002954951,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.003250445,
        "budget_committed_upper_usd": 0.003250445,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 8,
        "median": 5.2170065,
        "p90": 28.976971,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 8,
        "median": 5.2170065,
        "p90": 28.976971,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 8,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 8,
        "median": 5.2107145,
        "p90": 28.972651000000003,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "T",
      "stratum": "listed-control-SDNT",
      "scheduled_decisions": 20,
      "distinct_companies": 20,
      "outcomes": {
        "correct": 20,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 20
      },
      "eligible_membership_decisions": 20,
      "full_visible_answers_reviewed": 20,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 20
      },
      "recorded_paid_http_attempts": 55,
      "recorded_api_turns": 55,
      "recorded_http_attempts_including_prior_capacity_rejections": 55,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 55,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 19
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.010141087,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.011155201,
        "budget_committed_upper_usd": 0.011155201,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 20,
        "median": 6.4427455,
        "p90": 8.394214,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 20,
        "median": 6.4427455,
        "p90": 8.394214,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 20,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 20,
        "median": 6.4325730000000005,
        "p90": 8.384719,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "T",
      "stratum": "listed-control-SDNTK",
      "scheduled_decisions": 4,
      "distinct_companies": 4,
      "outcomes": {
        "correct": 4,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 4
      },
      "eligible_membership_decisions": 4,
      "full_visible_answers_reviewed": 4,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 4
      },
      "recorded_paid_http_attempts": 8,
      "recorded_api_turns": 8,
      "recorded_http_attempts_including_prior_capacity_rejections": 10,
      "documented_uncharged_capacity_rejections": 2,
      "generated_response_attempts": 8,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 2,
        "tier_completion": 2
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.001226,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.001348601,
        "budget_committed_upper_usd": 0.001348601,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 4,
        "median": 12.334123499999999,
        "p90": 54.363538,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 4,
        "median": 12.334123499999999,
        "p90": 54.363538,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 4,
        "median": 0.0,
        "p90": 6,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 4,
        "median": 12.3257175,
        "p90": 48.34552699999999,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "T",
      "stratum": "removed-20260528",
      "scheduled_decisions": 9,
      "distinct_companies": 9,
      "outcomes": {
        "correct": 8,
        "wrong": 0,
        "cannot_verify": 1,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 9
      },
      "eligible_membership_decisions": 9,
      "full_visible_answers_reviewed": 9,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 9
      },
      "recorded_paid_http_attempts": 28,
      "recorded_api_turns": 28,
      "recorded_http_attempts_including_prior_capacity_rejections": 32,
      "documented_uncharged_capacity_rejections": 4,
      "generated_response_attempts": 28,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "original_followup": 1,
        "tier_completion": 7
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.005742958,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.006317255,
        "budget_committed_upper_usd": 0.006317255,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 9,
        "median": 9.95732,
        "p90": 85.106741,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 9,
        "median": 9.95732,
        "p90": 85.106741,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 9,
        "median": 0,
        "p90": 16,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 9,
        "median": 9.947773,
        "p90": 69.075153,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "T",
      "stratum": "removed-20260727",
      "scheduled_decisions": 27,
      "distinct_companies": 27,
      "outcomes": {
        "correct": 25,
        "wrong": 0,
        "cannot_verify": 2,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 27
      },
      "eligible_membership_decisions": 27,
      "full_visible_answers_reviewed": 27,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 27
      },
      "recorded_paid_http_attempts": 61,
      "recorded_api_turns": 61,
      "recorded_http_attempts_including_prior_capacity_rejections": 61,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 61,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 26
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.014615675,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.016077244,
        "budget_committed_upper_usd": 0.016077244,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 27,
        "median": 7.548191,
        "p90": 10.340998,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 27,
        "median": 7.548191,
        "p90": 10.340998,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 27,
        "median": 0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 27,
        "median": 7.54148,
        "p90": 10.331968999999999,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "T",
      "stratum": "removed-20261005",
      "scheduled_decisions": 28,
      "distinct_companies": 28,
      "outcomes": {
        "correct": 26,
        "wrong": 0,
        "cannot_verify": 2,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 28
      },
      "eligible_membership_decisions": 28,
      "full_visible_answers_reviewed": 28,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 26,
        "query_identity_error": 2
      },
      "recorded_paid_http_attempts": 78,
      "recorded_api_turns": 78,
      "recorded_http_attempts_including_prior_capacity_rejections": 79,
      "documented_uncharged_capacity_rejections": 1,
      "generated_response_attempts": 78,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 27
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.017675745,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.019443324,
        "budget_committed_upper_usd": 0.019443324,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 28,
        "median": 8.8942105,
        "p90": 11.461684,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 28,
        "median": 8.8942105,
        "p90": 11.461684,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 28,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 28,
        "median": 8.885154499999999,
        "p90": 11.452494000000002,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-high",
      "stratum": "listed-control-IRAQ2",
      "scheduled_decisions": 8,
      "distinct_companies": 8,
      "outcomes": {
        "correct": 3,
        "wrong": 0,
        "cannot_verify": 5,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 8
      },
      "eligible_membership_decisions": 8,
      "full_visible_answers_reviewed": 8,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 8
      },
      "recorded_paid_http_attempts": 8,
      "recorded_api_turns": 8,
      "recorded_http_attempts_including_prior_capacity_rejections": 13,
      "documented_uncharged_capacity_rejections": 5,
      "generated_response_attempts": 8,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 2,
        "tier_completion": 6
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 8,
          "median": 2.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 8,
          "median": 2.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 8,
          "median": 1.0,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.13311432,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.146425753,
        "budget_committed_upper_usd": 0.146425753,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 8,
        "median": 13.8422585,
        "p90": 136.902643,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 8,
        "median": 13.8422585,
        "p90": 136.902643,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 8,
        "median": 0.0,
        "p90": 28,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 13.369162,
        "p90": 13.369162,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 8,
        "median": 13.836684,
        "p90": 108.87952200000001,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-high",
      "stratum": "listed-control-SDNT",
      "scheduled_decisions": 20,
      "distinct_companies": 20,
      "outcomes": {
        "correct": 11,
        "wrong": 0,
        "cannot_verify": 9,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 20
      },
      "eligible_membership_decisions": 20,
      "full_visible_answers_reviewed": 20,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 20
      },
      "recorded_paid_http_attempts": 20,
      "recorded_api_turns": 20,
      "recorded_http_attempts_including_prior_capacity_rejections": 26,
      "documented_uncharged_capacity_rejections": 6,
      "generated_response_attempts": 20,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 19
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 20,
          "median": 4.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 20,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 20,
          "median": 2.0,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.37054041,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.451594454,
        "budget_committed_upper_usd": 0.451594454,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 20,
        "median": 17.243733499999998,
        "p90": 23.020211,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 20,
        "median": 17.243733499999998,
        "p90": 23.020211,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 20,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 20,
        "median": 17.239549500000003,
        "p90": 23.014961,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-high",
      "stratum": "listed-control-SDNTK",
      "scheduled_decisions": 4,
      "distinct_companies": 4,
      "outcomes": {
        "correct": 3,
        "wrong": 0,
        "cannot_verify": 1,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 4
      },
      "eligible_membership_decisions": 4,
      "full_visible_answers_reviewed": 4,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 4
      },
      "recorded_paid_http_attempts": 4,
      "recorded_api_turns": 4,
      "recorded_http_attempts_including_prior_capacity_rejections": 14,
      "documented_uncharged_capacity_rejections": 10,
      "generated_response_attempts": 4,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 2,
        "tier_completion": 2
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 4,
          "median": 2.5,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 4,
          "median": 2.5,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 4,
          "median": 1.5,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.066353985,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.072989384,
        "budget_committed_upper_usd": 0.072989384,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 4,
        "median": 92.7395825,
        "p90": 214.111709,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 4,
        "median": 92.7395825,
        "p90": 214.111709,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 4,
        "median": 30.0,
        "p90": 60,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 4,
        "median": 62.7245645,
        "p90": 154.080958,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-high",
      "stratum": "removed-20260528",
      "scheduled_decisions": 9,
      "distinct_companies": 9,
      "outcomes": {
        "correct": 0,
        "wrong": 1,
        "cannot_verify": 8,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 9
      },
      "eligible_membership_decisions": 9,
      "full_visible_answers_reviewed": 9,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 8,
        "unclear": 1
      },
      "recorded_paid_http_attempts": 9,
      "recorded_api_turns": 9,
      "recorded_http_attempts_including_prior_capacity_rejections": 29,
      "documented_uncharged_capacity_rejections": 20,
      "generated_response_attempts": 9,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 8
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 9,
          "median": 5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 9,
          "median": 4,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 9,
          "median": 2,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.160925375,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.210017913,
        "budget_committed_upper_usd": 0.210017913,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 9,
        "median": 17.644619,
        "p90": 367.933403,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 9,
        "median": 17.644619,
        "p90": 367.933403,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 9,
        "median": 0,
        "p90": 150,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 212.670827,
        "p90": 212.670827,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 9,
        "median": 17.607922,
        "p90": 217.878313,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-high",
      "stratum": "removed-20260727",
      "scheduled_decisions": 27,
      "distinct_companies": 27,
      "outcomes": {
        "correct": 0,
        "wrong": 4,
        "cannot_verify": 23,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 27
      },
      "eligible_membership_decisions": 27,
      "full_visible_answers_reviewed": 27,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 23,
        "unclear": 4
      },
      "recorded_paid_http_attempts": 27,
      "recorded_api_turns": 27,
      "recorded_http_attempts_including_prior_capacity_rejections": 29,
      "documented_uncharged_capacity_rejections": 2,
      "generated_response_attempts": 27,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 26
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 27,
          "median": 5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 27,
          "median": 4,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 27,
          "median": 2,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.430416975,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.594458675,
        "budget_committed_upper_usd": 0.594458675,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 27,
        "median": 23.672006,
        "p90": 31.38845,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 27,
        "median": 23.672006,
        "p90": 31.38845,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 27,
        "median": 0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 27,
        "median": 23.666382,
        "p90": 31.382632,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-high",
      "stratum": "removed-20261005",
      "scheduled_decisions": 28,
      "distinct_companies": 28,
      "outcomes": {
        "correct": 4,
        "wrong": 9,
        "cannot_verify": 15,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 28
      },
      "eligible_membership_decisions": 28,
      "full_visible_answers_reviewed": 28,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 19,
        "unclear": 9
      },
      "recorded_paid_http_attempts": 28,
      "recorded_api_turns": 28,
      "recorded_http_attempts_including_prior_capacity_rejections": 40,
      "documented_uncharged_capacity_rejections": 12,
      "generated_response_attempts": 28,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "tier_completion": 28
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 28,
          "median": 5.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 28,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 28,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.48163674,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.573800418,
        "budget_committed_upper_usd": 0.573800418,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 28,
        "median": 22.397630499999998,
        "p90": 27.050771,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 28,
        "median": 22.397630499999998,
        "p90": 27.050771,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 28,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 308.388281,
        "p90": 308.388281,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 28,
        "median": 22.3928565,
        "p90": 27.045233,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-low",
      "stratum": "listed-control-IRAQ2",
      "scheduled_decisions": 8,
      "distinct_companies": 8,
      "outcomes": {
        "correct": 3,
        "wrong": 0,
        "cannot_verify": 5,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 8
      },
      "eligible_membership_decisions": 8,
      "full_visible_answers_reviewed": 8,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 3,
        "unclear": 5
      },
      "recorded_paid_http_attempts": 8,
      "recorded_api_turns": 8,
      "recorded_http_attempts_including_prior_capacity_rejections": 9,
      "documented_uncharged_capacity_rejections": 1,
      "generated_response_attempts": 8,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 2,
        "tier_completion": 6
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 8,
          "median": 4.5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 8,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 8,
          "median": 1.5,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.104909553,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.148400508,
        "budget_committed_upper_usd": 0.148400508,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 8,
        "median": 17.371915,
        "p90": 38.74008,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 8,
        "median": 17.371915,
        "p90": 38.74008,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 8,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 54.000262,
        "p90": 54.000262,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 8,
        "median": 17.3666215,
        "p90": 38.736224,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-low",
      "stratum": "listed-control-SDNT",
      "scheduled_decisions": 20,
      "distinct_companies": 20,
      "outcomes": {
        "correct": 14,
        "wrong": 0,
        "cannot_verify": 6,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 20
      },
      "eligible_membership_decisions": 20,
      "full_visible_answers_reviewed": 20,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 14,
        "unclear": 6
      },
      "recorded_paid_http_attempts": 20,
      "recorded_api_turns": 20,
      "recorded_http_attempts_including_prior_capacity_rejections": 22,
      "documented_uncharged_capacity_rejections": 2,
      "generated_response_attempts": 20,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 19
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 20,
          "median": 3.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 20,
          "median": 3.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 20,
          "median": 2.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.33891518,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.416806699,
        "budget_committed_upper_usd": 0.416806699,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 20,
        "median": 14.6003675,
        "p90": 19.754034,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 20,
        "median": 14.6003675,
        "p90": 19.754034,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 20,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 20,
        "median": 14.5955795,
        "p90": 19.749145,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-low",
      "stratum": "listed-control-SDNTK",
      "scheduled_decisions": 4,
      "distinct_companies": 4,
      "outcomes": {
        "correct": 1,
        "wrong": 0,
        "cannot_verify": 3,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 4
      },
      "eligible_membership_decisions": 4,
      "full_visible_answers_reviewed": 4,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 1,
        "unclear": 3
      },
      "recorded_paid_http_attempts": 4,
      "recorded_api_turns": 4,
      "recorded_http_attempts_including_prior_capacity_rejections": 7,
      "documented_uncharged_capacity_rejections": 3,
      "generated_response_attempts": 4,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 3
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 4,
          "median": 4.5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 4,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 4,
          "median": 2.0,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.07620457,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.094825028,
        "budget_committed_upper_usd": 0.094825028,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 4,
        "median": 19.4454415,
        "p90": 107.588181,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 4,
        "median": 19.4454415,
        "p90": 107.588181,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 4,
        "median": 0.0,
        "p90": 14,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 4,
        "median": 19.4407365,
        "p90": 93.56525099999999,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-low",
      "stratum": "removed-20260528",
      "scheduled_decisions": 9,
      "distinct_companies": 9,
      "outcomes": {
        "correct": 0,
        "wrong": 0,
        "cannot_verify": 9,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 9
      },
      "eligible_membership_decisions": 9,
      "full_visible_answers_reviewed": 9,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "unclear": 9
      },
      "recorded_paid_http_attempts": 9,
      "recorded_api_turns": 9,
      "recorded_http_attempts_including_prior_capacity_rejections": 21,
      "documented_uncharged_capacity_rejections": 12,
      "generated_response_attempts": 9,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 2,
        "tier_completion": 7
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 9,
          "median": 5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 9,
          "median": 4,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 9,
          "median": 2,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.150313245,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.19834457,
        "budget_committed_upper_usd": 0.19834457,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 9,
        "median": 19.08842,
        "p90": 366.537828,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 9,
        "median": 19.08842,
        "p90": 366.537828,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 9,
        "median": 0,
        "p90": 208,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 40.886681,
        "p90": 40.886681,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 9,
        "median": 19.082819,
        "p90": 158.490072,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-low",
      "stratum": "removed-20260727",
      "scheduled_decisions": 27,
      "distinct_companies": 27,
      "outcomes": {
        "correct": 0,
        "wrong": 1,
        "cannot_verify": 26,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 27
      },
      "eligible_membership_decisions": 27,
      "full_visible_answers_reviewed": 27,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "unclear": 27
      },
      "recorded_paid_http_attempts": 27,
      "recorded_api_turns": 27,
      "recorded_http_attempts_including_prior_capacity_rejections": 28,
      "documented_uncharged_capacity_rejections": 1,
      "generated_response_attempts": 27,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 26
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 27,
          "median": 5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 27,
          "median": 4,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 27,
          "median": 2,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.51547243,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.611019677,
        "budget_committed_upper_usd": 0.611019677,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 27,
        "median": 21.46767,
        "p90": 29.137246,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 27,
        "median": 21.46767,
        "p90": 29.137246,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 27,
        "median": 0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 27,
        "median": 21.462667,
        "p90": 29.131472,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-low",
      "stratum": "removed-20261005",
      "scheduled_decisions": 28,
      "distinct_companies": 28,
      "outcomes": {
        "correct": 4,
        "wrong": 7,
        "cannot_verify": 17,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 28
      },
      "eligible_membership_decisions": 28,
      "full_visible_answers_reviewed": 28,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 4,
        "unclear": 24
      },
      "recorded_paid_http_attempts": 28,
      "recorded_api_turns": 28,
      "recorded_http_attempts_including_prior_capacity_rejections": 28,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 28,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 27
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 28,
          "median": 5.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 28,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 28,
          "median": 1.5,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.468770937,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.570648032,
        "budget_committed_upper_usd": 0.570648032,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 28,
        "median": 21.886344,
        "p90": 30.194231,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 28,
        "median": 21.886344,
        "p90": 30.194231,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 28,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 28,
        "median": 21.8809995,
        "p90": 30.188456,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W8-low",
      "stratum": "listed-control-IRAQ2",
      "scheduled_decisions": 8,
      "distinct_companies": 8,
      "outcomes": {
        "correct": 7,
        "wrong": 0,
        "cannot_verify": 1,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 8
      },
      "eligible_membership_decisions": 8,
      "full_visible_answers_reviewed": 8,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 8
      },
      "recorded_paid_http_attempts": 8,
      "recorded_api_turns": 8,
      "recorded_http_attempts_including_prior_capacity_rejections": 12,
      "documented_uncharged_capacity_rejections": 4,
      "generated_response_attempts": 8,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 2,
        "tier_completion": 6
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 8,
          "median": 1.5,
          "p90": 6,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 8,
          "median": 1.5,
          "p90": 6,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 8,
          "median": 1.5,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.143977625,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.158375388,
        "budget_committed_upper_usd": 0.158375388,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 8,
        "median": 18.895473,
        "p90": 118.230955,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 8,
        "median": 18.895473,
        "p90": 118.230955,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 8,
        "median": 0.0,
        "p90": 28,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 55.397794,
        "p90": 55.397794,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 8,
        "median": 18.8911265,
        "p90": 90.210675,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W8-low",
      "stratum": "listed-control-SDNT",
      "scheduled_decisions": 20,
      "distinct_companies": 20,
      "outcomes": {
        "correct": 14,
        "wrong": 0,
        "cannot_verify": 6,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 20
      },
      "eligible_membership_decisions": 20,
      "full_visible_answers_reviewed": 20,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 20
      },
      "recorded_paid_http_attempts": 20,
      "recorded_api_turns": 20,
      "recorded_http_attempts_including_prior_capacity_rejections": 20,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 20,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 19
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 20,
          "median": 3.0,
          "p90": 6,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 20,
          "median": 3.0,
          "p90": 6,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 20,
          "median": 2.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.426987515,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.469686267,
        "budget_committed_upper_usd": 0.469686267,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 20,
        "median": 14.6128,
        "p90": 25.569013,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 20,
        "median": 14.6128,
        "p90": 25.569013,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 20,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 20,
        "median": 14.609003999999999,
        "p90": 25.562997,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W8-low",
      "stratum": "listed-control-SDNTK",
      "scheduled_decisions": 4,
      "distinct_companies": 4,
      "outcomes": {
        "correct": 3,
        "wrong": 0,
        "cannot_verify": 1,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 4
      },
      "eligible_membership_decisions": 4,
      "full_visible_answers_reviewed": 4,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 4
      },
      "recorded_paid_http_attempts": 4,
      "recorded_api_turns": 4,
      "recorded_http_attempts_including_prior_capacity_rejections": 21,
      "documented_uncharged_capacity_rejections": 17,
      "generated_response_attempts": 4,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 3
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 4,
          "median": 2.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 4,
          "median": 2.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 4,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.05553553,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.061089083,
        "budget_committed_upper_usd": 0.061089083,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 4,
        "median": 12.884518,
        "p90": 180.772197,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 4,
        "median": 12.884518,
        "p90": 180.772197,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 4,
        "median": 0.0,
        "p90": 60,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 283.79864200000003,
        "p90": 283.79864200000003,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 4,
        "median": 12.8779655,
        "p90": 120.73906100000002,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W8-low",
      "stratum": "removed-20260528",
      "scheduled_decisions": 9,
      "distinct_companies": 9,
      "outcomes": {
        "correct": 1,
        "wrong": 0,
        "cannot_verify": 8,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 9
      },
      "eligible_membership_decisions": 9,
      "full_visible_answers_reviewed": 9,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 9
      },
      "recorded_paid_http_attempts": 9,
      "recorded_api_turns": 9,
      "recorded_http_attempts_including_prior_capacity_rejections": 23,
      "documented_uncharged_capacity_rejections": 14,
      "generated_response_attempts": 9,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 2,
        "tier_completion": 7
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 9,
          "median": 5,
          "p90": 8,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 9,
          "median": 5,
          "p90": 8,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 9,
          "median": 2,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.235634945,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.25919844,
        "budget_committed_upper_usd": 0.25919844,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 9,
        "median": 25.048006,
        "p90": 723.731584,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 9,
        "median": 25.048006,
        "p90": 723.731584,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 9,
        "median": 0,
        "p90": 240,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 16.779046,
        "p90": 16.779046,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 9,
        "median": 25.044745,
        "p90": 483.63006599999994,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W8-low",
      "stratum": "removed-20260727",
      "scheduled_decisions": 27,
      "distinct_companies": 27,
      "outcomes": {
        "correct": 0,
        "wrong": 3,
        "cannot_verify": 22,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 2,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 25,
        "truncated_output": 2
      },
      "eligible_membership_decisions": 25,
      "full_visible_answers_reviewed": 26,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 1,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 23,
        "unclear": 3
      },
      "recorded_paid_http_attempts": 27,
      "recorded_api_turns": 27,
      "recorded_http_attempts_including_prior_capacity_rejections": 39,
      "documented_uncharged_capacity_rejections": 12,
      "generated_response_attempts": 27,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "tier_completion": 27
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 27,
          "median": 8,
          "p90": 9,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 27,
          "median": 8,
          "p90": 8,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 27,
          "median": 2,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.728538395,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.823392236,
        "budget_committed_upper_usd": 0.823392236,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 27,
        "median": 29.395295,
        "p90": 43.030033,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 25,
        "median": 28.598001,
        "p90": 40.932209,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 27,
        "median": 0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 357.0459649999999,
        "p90": 357.0459649999999,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 27,
        "median": 29.392057,
        "p90": 43.023169,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W8-low",
      "stratum": "removed-20261005",
      "scheduled_decisions": 28,
      "distinct_companies": 28,
      "outcomes": {
        "correct": 10,
        "wrong": 7,
        "cannot_verify": 11,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 28
      },
      "eligible_membership_decisions": 28,
      "full_visible_answers_reviewed": 28,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 21,
        "unclear": 7
      },
      "recorded_paid_http_attempts": 28,
      "recorded_api_turns": 28,
      "recorded_http_attempts_including_prior_capacity_rejections": 40,
      "documented_uncharged_capacity_rejections": 12,
      "generated_response_attempts": 28,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "tier_completion": 28
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 28,
          "median": 5.0,
          "p90": 9,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 28,
          "median": 5.0,
          "p90": 8,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 28,
          "median": 2.0,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.6198797,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.71486767,
        "budget_committed_upper_usd": 0.71486767,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 28,
        "median": 19.825941,
        "p90": 33.171036,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 28,
        "median": 19.825941,
        "p90": 33.171036,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 28,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 267.94633600000003,
        "p90": 267.94633600000003,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 28,
        "median": 19.821778000000002,
        "p90": 33.167323,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "D",
      "stratum": "listed-control-IRAQ2",
      "scheduled_decisions": 8,
      "distinct_companies": 8,
      "outcomes": {
        "correct": 8,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 8
      },
      "eligible_membership_decisions": 8,
      "full_visible_answers_reviewed": 8,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 8
      },
      "recorded_paid_http_attempts": 8,
      "recorded_api_turns": 8,
      "recorded_http_attempts_including_prior_capacity_rejections": 9,
      "documented_uncharged_capacity_rejections": 1,
      "generated_response_attempts": 8,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "original_followup": 1,
        "tier_completion": 6
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.018879,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.0207669,
        "budget_committed_upper_usd": 0.0207669,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 8,
        "median": 12.037033000000001,
        "p90": 15.891546,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 8,
        "median": 12.037033000000001,
        "p90": 15.891546,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 8,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 8,
        "median": 11.6498125,
        "p90": 15.887192,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "D",
      "stratum": "listed-control-SDNT",
      "scheduled_decisions": 20,
      "distinct_companies": 20,
      "outcomes": {
        "correct": 20,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 20
      },
      "eligible_membership_decisions": 20,
      "full_visible_answers_reviewed": 20,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 20
      },
      "recorded_paid_http_attempts": 20,
      "recorded_api_turns": 20,
      "recorded_http_attempts_including_prior_capacity_rejections": 24,
      "documented_uncharged_capacity_rejections": 4,
      "generated_response_attempts": 20,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 19
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.0484717,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.05331887,
        "budget_committed_upper_usd": 0.05331887,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 20,
        "median": 12.25281,
        "p90": 15.516477,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 20,
        "median": 12.25281,
        "p90": 15.516477,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 20,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 20,
        "median": 12.046308499999999,
        "p90": 13.503425,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "D",
      "stratum": "listed-control-SDNTK",
      "scheduled_decisions": 4,
      "distinct_companies": 4,
      "outcomes": {
        "correct": 4,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 4
      },
      "eligible_membership_decisions": 4,
      "full_visible_answers_reviewed": 4,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 4
      },
      "recorded_paid_http_attempts": 4,
      "recorded_api_turns": 4,
      "recorded_http_attempts_including_prior_capacity_rejections": 4,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 4,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 3
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.009842,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.0108262,
        "budget_committed_upper_usd": 0.0108262,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 4,
        "median": 11.734316,
        "p90": 12.054782,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 4,
        "median": 11.734316,
        "p90": 12.054782,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 4,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 4,
        "median": 11.7300685,
        "p90": 12.05212,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "D",
      "stratum": "removed-20260528",
      "scheduled_decisions": 9,
      "distinct_companies": 9,
      "outcomes": {
        "correct": 0,
        "wrong": 0,
        "cannot_verify": 9,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 9
      },
      "eligible_membership_decisions": 9,
      "full_visible_answers_reviewed": 9,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 9
      },
      "recorded_paid_http_attempts": 9,
      "recorded_api_turns": 9,
      "recorded_http_attempts_including_prior_capacity_rejections": 9,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 9,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "original_followup": 1,
        "tier_completion": 7
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.025367,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.0279037,
        "budget_committed_upper_usd": 0.0279037,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 9,
        "median": 13.977399,
        "p90": 17.869179,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 9,
        "median": 13.977399,
        "p90": 17.869179,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 9,
        "median": 0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 9,
        "median": 13.973411,
        "p90": 17.863167,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "D",
      "stratum": "removed-20260727",
      "scheduled_decisions": 27,
      "distinct_companies": 27,
      "outcomes": {
        "correct": 5,
        "wrong": 0,
        "cannot_verify": 22,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 27
      },
      "eligible_membership_decisions": 27,
      "full_visible_answers_reviewed": 27,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 27
      },
      "recorded_paid_http_attempts": 27,
      "recorded_api_turns": 27,
      "recorded_http_attempts_including_prior_capacity_rejections": 33,
      "documented_uncharged_capacity_rejections": 6,
      "generated_response_attempts": 27,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 26
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.088165,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.0969815,
        "budget_committed_upper_usd": 0.0969815,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 27,
        "median": 15.336705,
        "p90": 18.970514,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 27,
        "median": 15.336705,
        "p90": 18.970514,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 27,
        "median": 0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 27,
        "median": 15.222662999999999,
        "p90": 18.60197,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "D",
      "stratum": "removed-20261005",
      "scheduled_decisions": 28,
      "distinct_companies": 28,
      "outcomes": {
        "correct": 28,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 28
      },
      "eligible_membership_decisions": 28,
      "full_visible_answers_reviewed": 28,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 28
      },
      "recorded_paid_http_attempts": 28,
      "recorded_api_turns": 28,
      "recorded_http_attempts_including_prior_capacity_rejections": 32,
      "documented_uncharged_capacity_rejections": 4,
      "generated_response_attempts": 28,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 27
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.08634795,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.094982745,
        "budget_committed_upper_usd": 0.094982745,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 28,
        "median": 14.539689,
        "p90": 18.930985,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 28,
        "median": 14.539689,
        "p90": 18.930985,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 28,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 28,
        "median": 14.0572695,
        "p90": 17.716105,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "T",
      "stratum": "listed-control-IRAQ2",
      "scheduled_decisions": 8,
      "distinct_companies": 8,
      "outcomes": {
        "correct": 8,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 8
      },
      "eligible_membership_decisions": 8,
      "full_visible_answers_reviewed": 8,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 8
      },
      "recorded_paid_http_attempts": 16,
      "recorded_api_turns": 16,
      "recorded_http_attempts_including_prior_capacity_rejections": 16,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 16,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "original_followup": 1,
        "tier_completion": 6
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.02845975,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.031305725,
        "budget_committed_upper_usd": 0.031305725,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 8,
        "median": 10.7432525,
        "p90": 17.492304,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 8,
        "median": 10.7432525,
        "p90": 17.492304,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 8,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 8,
        "median": 10.7357805,
        "p90": 17.489530000000002,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "T",
      "stratum": "listed-control-SDNT",
      "scheduled_decisions": 20,
      "distinct_companies": 20,
      "outcomes": {
        "correct": 20,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 20
      },
      "eligible_membership_decisions": 20,
      "full_visible_answers_reviewed": 20,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 20
      },
      "recorded_paid_http_attempts": 40,
      "recorded_api_turns": 40,
      "recorded_http_attempts_including_prior_capacity_rejections": 45,
      "documented_uncharged_capacity_rejections": 5,
      "generated_response_attempts": 40,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 19
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.07172555,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.078898105,
        "budget_committed_upper_usd": 0.078898105,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 20,
        "median": 14.5329095,
        "p90": 24.267432,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 20,
        "median": 14.5329095,
        "p90": 24.267432,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 20,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 20,
        "median": 14.0110855,
        "p90": 22.252412,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "T",
      "stratum": "listed-control-SDNTK",
      "scheduled_decisions": 4,
      "distinct_companies": 4,
      "outcomes": {
        "correct": 4,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 4
      },
      "eligible_membership_decisions": 4,
      "full_visible_answers_reviewed": 4,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 4
      },
      "recorded_paid_http_attempts": 8,
      "recorded_api_turns": 8,
      "recorded_http_attempts_including_prior_capacity_rejections": 8,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 8,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 3
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.01473675,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.016210425,
        "budget_committed_upper_usd": 0.016210425,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 4,
        "median": 10.8044765,
        "p90": 12.387509,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 4,
        "median": 10.8044765,
        "p90": 12.387509,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 4,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 4,
        "median": 10.79795,
        "p90": 12.378734999999999,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "T",
      "stratum": "removed-20260528",
      "scheduled_decisions": 9,
      "distinct_companies": 9,
      "outcomes": {
        "correct": 9,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 9
      },
      "eligible_membership_decisions": 9,
      "full_visible_answers_reviewed": 9,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 9
      },
      "recorded_paid_http_attempts": 27,
      "recorded_api_turns": 27,
      "recorded_http_attempts_including_prior_capacity_rejections": 27,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 27,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "original_followup": 1,
        "tier_completion": 7
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.06061125,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.066672375,
        "budget_committed_upper_usd": 0.066672375,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 9,
        "median": 24.269033,
        "p90": 34.039638,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 9,
        "median": 24.269033,
        "p90": 34.039638,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 9,
        "median": 0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 9,
        "median": 24.261409,
        "p90": 34.030184,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "T",
      "stratum": "removed-20260727",
      "scheduled_decisions": 27,
      "distinct_companies": 27,
      "outcomes": {
        "correct": 26,
        "wrong": 0,
        "cannot_verify": 1,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 27
      },
      "eligible_membership_decisions": 27,
      "full_visible_answers_reviewed": 27,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 27
      },
      "recorded_paid_http_attempts": 73,
      "recorded_api_turns": 73,
      "recorded_http_attempts_including_prior_capacity_rejections": 82,
      "documented_uncharged_capacity_rejections": 9,
      "generated_response_attempts": 73,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 26
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.1444461,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.15889071,
        "budget_committed_upper_usd": 0.15889071,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 27,
        "median": 19.150274,
        "p90": 32.628246,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 27,
        "median": 19.150274,
        "p90": 32.628246,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 27,
        "median": 0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 27,
        "median": 18.68741,
        "p90": 30.6162,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "T",
      "stratum": "removed-20261005",
      "scheduled_decisions": 28,
      "distinct_companies": 28,
      "outcomes": {
        "correct": 28,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 28
      },
      "eligible_membership_decisions": 28,
      "full_visible_answers_reviewed": 28,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 28
      },
      "recorded_paid_http_attempts": 77,
      "recorded_api_turns": 77,
      "recorded_http_attempts_including_prior_capacity_rejections": 83,
      "documented_uncharged_capacity_rejections": 6,
      "generated_response_attempts": 77,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 27
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.1667209,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.18339299,
        "budget_committed_upper_usd": 0.18339299,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 28,
        "median": 17.977997000000002,
        "p90": 34.699877,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 28,
        "median": 17.977997000000002,
        "p90": 34.699877,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 28,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 28,
        "median": 17.9677925,
        "p90": 31.814814,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-high",
      "stratum": "listed-control-IRAQ2",
      "scheduled_decisions": 8,
      "distinct_companies": 8,
      "outcomes": {
        "correct": 7,
        "wrong": 0,
        "cannot_verify": 1,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 8
      },
      "eligible_membership_decisions": 8,
      "full_visible_answers_reviewed": 8,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 8
      },
      "recorded_paid_http_attempts": 8,
      "recorded_api_turns": 8,
      "recorded_http_attempts_including_prior_capacity_rejections": 10,
      "documented_uncharged_capacity_rejections": 2,
      "generated_response_attempts": 8,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 2,
        "tier_completion": 6
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 8,
          "median": 2.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 8,
          "median": 2.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 8,
          "median": 1.0,
          "p90": 1,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.18632055,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.204952605,
        "budget_committed_upper_usd": 0.204952605,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 8,
        "median": 20.207399,
        "p90": 39.169263,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 8,
        "median": 20.207399,
        "p90": 39.169263,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 8,
        "median": 0.0,
        "p90": 4,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 11.049299,
        "p90": 11.049299,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 8,
        "median": 20.202832,
        "p90": 39.165999,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-high",
      "stratum": "listed-control-SDNT",
      "scheduled_decisions": 20,
      "distinct_companies": 20,
      "outcomes": {
        "correct": 17,
        "wrong": 0,
        "cannot_verify": 3,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 20
      },
      "eligible_membership_decisions": 20,
      "full_visible_answers_reviewed": 20,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 20
      },
      "recorded_paid_http_attempts": 20,
      "recorded_api_turns": 20,
      "recorded_http_attempts_including_prior_capacity_rejections": 27,
      "documented_uncharged_capacity_rejections": 7,
      "generated_response_attempts": 20,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 19
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 20,
          "median": 3.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 20,
          "median": 3.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 20,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.5547897,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.61026867,
        "budget_committed_upper_usd": 0.61026867,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 20,
        "median": 32.554521,
        "p90": 48.775762,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 20,
        "median": 32.554521,
        "p90": 48.775762,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 20,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 20,
        "median": 31.9890035,
        "p90": 46.76561,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-high",
      "stratum": "listed-control-SDNTK",
      "scheduled_decisions": 4,
      "distinct_companies": 4,
      "outcomes": {
        "correct": 4,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 4
      },
      "eligible_membership_decisions": 4,
      "full_visible_answers_reviewed": 4,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 4
      },
      "recorded_paid_http_attempts": 4,
      "recorded_api_turns": 4,
      "recorded_http_attempts_including_prior_capacity_rejections": 4,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 4,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 2,
        "tier_completion": 2
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 4,
          "median": 2.5,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 4,
          "median": 2.5,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 4,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.10755715,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.118312865,
        "budget_committed_upper_usd": 0.118312865,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 4,
        "median": 27.1621305,
        "p90": 44.448006,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 4,
        "median": 27.1621305,
        "p90": 44.448006,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 4,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 4,
        "median": 27.157955,
        "p90": 44.445991,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-high",
      "stratum": "removed-20260528",
      "scheduled_decisions": 9,
      "distinct_companies": 9,
      "outcomes": {
        "correct": 0,
        "wrong": 0,
        "cannot_verify": 9,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 9
      },
      "eligible_membership_decisions": 9,
      "full_visible_answers_reviewed": 9,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 9
      },
      "recorded_paid_http_attempts": 9,
      "recorded_api_turns": 9,
      "recorded_http_attempts_including_prior_capacity_rejections": 12,
      "documented_uncharged_capacity_rejections": 3,
      "generated_response_attempts": 9,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 2,
        "tier_completion": 7
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 9,
          "median": 5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 9,
          "median": 4,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 9,
          "median": 1,
          "p90": 1,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.32909645,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.362006095,
        "budget_committed_upper_usd": 0.362006095,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 9,
        "median": 43.753732,
        "p90": 93.83099,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 9,
        "median": 43.753732,
        "p90": 93.83099,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 9,
        "median": 0,
        "p90": 14,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 9,
        "median": 43.748127,
        "p90": 79.807165,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-high",
      "stratum": "removed-20260727",
      "scheduled_decisions": 27,
      "distinct_companies": 27,
      "outcomes": {
        "correct": 0,
        "wrong": 0,
        "cannot_verify": 27,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 27
      },
      "eligible_membership_decisions": 27,
      "full_visible_answers_reviewed": 27,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 27
      },
      "recorded_paid_http_attempts": 27,
      "recorded_api_turns": 27,
      "recorded_http_attempts_including_prior_capacity_rejections": 44,
      "documented_uncharged_capacity_rejections": 17,
      "generated_response_attempts": 27,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 26
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 27,
          "median": 5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 27,
          "median": 4,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 27,
          "median": 1,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 1.139397,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 1.2643367,
        "budget_committed_upper_usd": 1.2643367,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 27,
        "median": 47.844562,
        "p90": 87.914627,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 27,
        "median": 47.844562,
        "p90": 87.914627,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 27,
        "median": 0,
        "p90": 6,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 27,
        "median": 47.841252,
        "p90": 81.890897,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-high",
      "stratum": "removed-20261005",
      "scheduled_decisions": 28,
      "distinct_companies": 28,
      "outcomes": {
        "correct": 18,
        "wrong": 0,
        "cannot_verify": 10,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 28
      },
      "eligible_membership_decisions": 28,
      "full_visible_answers_reviewed": 28,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 28
      },
      "recorded_paid_http_attempts": 28,
      "recorded_api_turns": 28,
      "recorded_http_attempts_including_prior_capacity_rejections": 43,
      "documented_uncharged_capacity_rejections": 15,
      "generated_response_attempts": 28,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 27
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 28,
          "median": 5.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 28,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 28,
          "median": 1.0,
          "p90": 1,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 1.24352355,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 1.378875905,
        "budget_committed_upper_usd": 1.378875905,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 28,
        "median": 51.048584,
        "p90": 97.89263,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 28,
        "median": 51.048584,
        "p90": 97.89263,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 28,
        "median": 0.0,
        "p90": 6,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 28,
        "median": 51.0443455,
        "p90": 91.87708900000001,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-low",
      "stratum": "listed-control-IRAQ2",
      "scheduled_decisions": 8,
      "distinct_companies": 8,
      "outcomes": {
        "correct": 6,
        "wrong": 0,
        "cannot_verify": 2,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 8
      },
      "eligible_membership_decisions": 8,
      "full_visible_answers_reviewed": 8,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 6,
        "unclear": 2
      },
      "recorded_paid_http_attempts": 8,
      "recorded_api_turns": 8,
      "recorded_http_attempts_including_prior_capacity_rejections": 8,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 8,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "original_followup": 1,
        "tier_completion": 6
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 8,
          "median": 2.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 8,
          "median": 2.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 8,
          "median": 1.0,
          "p90": 1,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.19122875,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.210351625,
        "budget_committed_upper_usd": 0.210351625,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 8,
        "median": 21.426978,
        "p90": 35.769518,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 8,
        "median": 21.426978,
        "p90": 35.769518,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 8,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 8,
        "median": 21.42172,
        "p90": 35.766224,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-low",
      "stratum": "listed-control-SDNT",
      "scheduled_decisions": 20,
      "distinct_companies": 20,
      "outcomes": {
        "correct": 13,
        "wrong": 0,
        "cannot_verify": 7,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 20
      },
      "eligible_membership_decisions": 20,
      "full_visible_answers_reviewed": 20,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 13,
        "unclear": 7
      },
      "recorded_paid_http_attempts": 20,
      "recorded_api_turns": 20,
      "recorded_http_attempts_including_prior_capacity_rejections": 28,
      "documented_uncharged_capacity_rejections": 8,
      "generated_response_attempts": 20,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 19
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 20,
          "median": 3.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 20,
          "median": 3.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 20,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.5814781,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.63962591,
        "budget_committed_upper_usd": 0.63962591,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 20,
        "median": 43.2838735,
        "p90": 57.944792,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 20,
        "median": 43.2838735,
        "p90": 57.944792,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 20,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 20,
        "median": 42.2262705,
        "p90": 57.939977,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-low",
      "stratum": "listed-control-SDNTK",
      "scheduled_decisions": 4,
      "distinct_companies": 4,
      "outcomes": {
        "correct": 4,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 4
      },
      "eligible_membership_decisions": 4,
      "full_visible_answers_reviewed": 4,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 4
      },
      "recorded_paid_http_attempts": 4,
      "recorded_api_turns": 4,
      "recorded_http_attempts_including_prior_capacity_rejections": 6,
      "documented_uncharged_capacity_rejections": 2,
      "generated_response_attempts": 4,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 2,
        "tier_completion": 2
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 4,
          "median": 2.5,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 4,
          "median": 2.5,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 4,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.10014855,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.110163405,
        "budget_committed_upper_usd": 0.110163405,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 4,
        "median": 20.698707499999998,
        "p90": 83.701051,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 4,
        "median": 20.698707499999998,
        "p90": 83.701051,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 4,
        "median": 0.0,
        "p90": 6,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 4,
        "median": 20.6949375,
        "p90": 77.68619000000001,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-low",
      "stratum": "removed-20260528",
      "scheduled_decisions": 9,
      "distinct_companies": 9,
      "outcomes": {
        "correct": 0,
        "wrong": 0,
        "cannot_verify": 9,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 9
      },
      "eligible_membership_decisions": 9,
      "full_visible_answers_reviewed": 9,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "unclear": 9
      },
      "recorded_paid_http_attempts": 9,
      "recorded_api_turns": 9,
      "recorded_http_attempts_including_prior_capacity_rejections": 10,
      "documented_uncharged_capacity_rejections": 1,
      "generated_response_attempts": 9,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 2,
        "tier_completion": 7
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 9,
          "median": 5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 9,
          "median": 4,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 9,
          "median": 1,
          "p90": 1,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.32043825,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.352482075,
        "budget_committed_upper_usd": 0.352482075,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 9,
        "median": 41.78071,
        "p90": 48.60646,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 9,
        "median": 41.78071,
        "p90": 48.60646,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 9,
        "median": 0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 9,
        "median": 41.77722,
        "p90": 48.601812,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-low",
      "stratum": "removed-20260727",
      "scheduled_decisions": 27,
      "distinct_companies": 27,
      "outcomes": {
        "correct": 1,
        "wrong": 0,
        "cannot_verify": 26,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 27
      },
      "eligible_membership_decisions": 27,
      "full_visible_answers_reviewed": 27,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 1,
        "unclear": 26
      },
      "recorded_paid_http_attempts": 27,
      "recorded_api_turns": 27,
      "recorded_http_attempts_including_prior_capacity_rejections": 34,
      "documented_uncharged_capacity_rejections": 7,
      "generated_response_attempts": 27,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 26
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 27,
          "median": 5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 27,
          "median": 4,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 27,
          "median": 1,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 1.1919506,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 1.34414566,
        "budget_committed_upper_usd": 1.34414566,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 27,
        "median": 42.626251,
        "p90": 75.317091,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 27,
        "median": 42.626251,
        "p90": 75.317091,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 27,
        "median": 0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 27,
        "median": 42.620025,
        "p90": 73.302651,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-low",
      "stratum": "removed-20261005",
      "scheduled_decisions": 28,
      "distinct_companies": 28,
      "outcomes": {
        "correct": 15,
        "wrong": 0,
        "cannot_verify": 13,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 28
      },
      "eligible_membership_decisions": 28,
      "full_visible_answers_reviewed": 28,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 15,
        "unclear": 13
      },
      "recorded_paid_http_attempts": 28,
      "recorded_api_turns": 28,
      "recorded_http_attempts_including_prior_capacity_rejections": 45,
      "documented_uncharged_capacity_rejections": 17,
      "generated_response_attempts": 28,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 27
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 28,
          "median": 5.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 28,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 28,
          "median": 1.0,
          "p90": 1,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 1.27040435,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 1.408444785,
        "budget_committed_upper_usd": 1.408444785,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 28,
        "median": 54.329745,
        "p90": 92.90252,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 28,
        "median": 54.329745,
        "p90": 92.90252,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 28,
        "median": 0.0,
        "p90": 6,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 28,
        "median": 54.0859095,
        "p90": 90.89095800000001,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W8-low",
      "stratum": "listed-control-IRAQ2",
      "scheduled_decisions": 8,
      "distinct_companies": 8,
      "outcomes": {
        "correct": 8,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 8
      },
      "eligible_membership_decisions": 8,
      "full_visible_answers_reviewed": 8,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 8
      },
      "recorded_paid_http_attempts": 8,
      "recorded_api_turns": 8,
      "recorded_http_attempts_including_prior_capacity_rejections": 10,
      "documented_uncharged_capacity_rejections": 2,
      "generated_response_attempts": 8,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 2,
        "tier_completion": 6
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 8,
          "median": 2.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 8,
          "median": 2.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 8,
          "median": 1.0,
          "p90": 1,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.18717555,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.205893105,
        "budget_committed_upper_usd": 0.205893105,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 8,
        "median": 23.989172,
        "p90": 38.246352,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 8,
        "median": 23.989172,
        "p90": 38.246352,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 8,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 11.064242,
        "p90": 11.064242,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 8,
        "median": 23.9857225,
        "p90": 38.239502,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W8-low",
      "stratum": "listed-control-SDNT",
      "scheduled_decisions": 20,
      "distinct_companies": 20,
      "outcomes": {
        "correct": 16,
        "wrong": 0,
        "cannot_verify": 4,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 20
      },
      "eligible_membership_decisions": 20,
      "full_visible_answers_reviewed": 20,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 16,
        "unclear": 4
      },
      "recorded_paid_http_attempts": 20,
      "recorded_api_turns": 20,
      "recorded_http_attempts_including_prior_capacity_rejections": 26,
      "documented_uncharged_capacity_rejections": 6,
      "generated_response_attempts": 20,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 19
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 20,
          "median": 2.5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 20,
          "median": 2.5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 20,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.5763997,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.63403967,
        "budget_committed_upper_usd": 0.63403967,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 20,
        "median": 30.9973575,
        "p90": 63.950416,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 20,
        "median": 30.9973575,
        "p90": 63.950416,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 20,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 20,
        "median": 30.9927685,
        "p90": 61.927854,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W8-low",
      "stratum": "listed-control-SDNTK",
      "scheduled_decisions": 4,
      "distinct_companies": 4,
      "outcomes": {
        "correct": 4,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 4
      },
      "eligible_membership_decisions": 4,
      "full_visible_answers_reviewed": 4,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 4
      },
      "recorded_paid_http_attempts": 4,
      "recorded_api_turns": 4,
      "recorded_http_attempts_including_prior_capacity_rejections": 4,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 4,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 3
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 4,
          "median": 2.5,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 4,
          "median": 2.5,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 4,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.11232335,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.123555685,
        "budget_committed_upper_usd": 0.123555685,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 4,
        "median": 15.508549,
        "p90": 42.443376,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 4,
        "median": 15.508549,
        "p90": 42.443376,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 4,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 4,
        "median": 15.504607,
        "p90": 42.438728,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W8-low",
      "stratum": "removed-20260528",
      "scheduled_decisions": 9,
      "distinct_companies": 9,
      "outcomes": {
        "correct": 0,
        "wrong": 0,
        "cannot_verify": 9,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 9
      },
      "eligible_membership_decisions": 9,
      "full_visible_answers_reviewed": 9,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 5,
        "unclear": 4
      },
      "recorded_paid_http_attempts": 9,
      "recorded_api_turns": 9,
      "recorded_http_attempts_including_prior_capacity_rejections": 13,
      "documented_uncharged_capacity_rejections": 4,
      "generated_response_attempts": 9,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 2,
        "tier_completion": 7
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 9,
          "median": 5,
          "p90": 6,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 9,
          "median": 5,
          "p90": 6,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 9,
          "median": 1,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.38446545,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.422911995,
        "budget_committed_upper_usd": 0.422911995,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 9,
        "median": 53.341505,
        "p90": 100.087125,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 9,
        "median": 53.341505,
        "p90": 100.087125,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 9,
        "median": 0,
        "p90": 6,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 9,
        "median": 53.336263,
        "p90": 94.058713,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W8-low",
      "stratum": "removed-20260727",
      "scheduled_decisions": 27,
      "distinct_companies": 27,
      "outcomes": {
        "correct": 1,
        "wrong": 0,
        "cannot_verify": 26,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 27
      },
      "eligible_membership_decisions": 27,
      "full_visible_answers_reviewed": 27,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 6,
        "unclear": 21
      },
      "recorded_paid_http_attempts": 27,
      "recorded_api_turns": 27,
      "recorded_http_attempts_including_prior_capacity_rejections": 45,
      "documented_uncharged_capacity_rejections": 18,
      "generated_response_attempts": 27,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 26
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 27,
          "median": 5,
          "p90": 6,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 27,
          "median": 5,
          "p90": 6,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 27,
          "median": 1,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 1.3546594,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 1.49012534,
        "budget_committed_upper_usd": 1.49012534,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 27,
        "median": 48.758248,
        "p90": 125.15736,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 27,
        "median": 48.758248,
        "p90": 125.15736,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 27,
        "median": 0,
        "p90": 14,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 27,
        "median": 48.752485,
        "p90": 108.55159499999999,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W8-low",
      "stratum": "removed-20261005",
      "scheduled_decisions": 28,
      "distinct_companies": 28,
      "outcomes": {
        "correct": 26,
        "wrong": 0,
        "cannot_verify": 2,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 28
      },
      "eligible_membership_decisions": 28,
      "full_visible_answers_reviewed": 28,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 26,
        "unclear": 2
      },
      "recorded_paid_http_attempts": 28,
      "recorded_api_turns": 28,
      "recorded_http_attempts_including_prior_capacity_rejections": 47,
      "documented_uncharged_capacity_rejections": 19,
      "generated_response_attempts": 28,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 1,
        "tier_completion": 27
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 28,
          "median": 5.0,
          "p90": 7,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 28,
          "median": 5.0,
          "p90": 7,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 28,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 1.55548315,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 1.711031465,
        "budget_committed_upper_usd": 1.711031465,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 28,
        "median": 50.8825535,
        "p90": 142.513968,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 28,
        "median": 50.8825535,
        "p90": 142.513968,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 28,
        "median": 0.0,
        "p90": 14,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 28,
        "median": 50.878614,
        "p90": 128.480651,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    }
  ],
  "by_execution_segment": [
    {
      "execution_segment": "capacity_amendment_concurrency8",
      "scheduled_decisions": 74,
      "distinct_companies": 9,
      "outcomes": {
        "correct": 46,
        "wrong": 1,
        "cannot_verify": 27,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 74
      },
      "eligible_membership_decisions": 74,
      "full_visible_answers_reviewed": 74,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 65,
        "unclear": 9
      },
      "recorded_paid_http_attempts": 93,
      "recorded_api_turns": 93,
      "recorded_http_attempts_including_prior_capacity_rejections": 195,
      "documented_uncharged_capacity_rejections": 102,
      "generated_response_attempts": 93,
      "decisions_with_prior_capacity_rejection": 9,
      "completed_after_prior_capacity_rejection": 9,
      "execution_phases": {
        "amendment": 74
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 46,
          "median": 3.5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 46,
          "median": 3.5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 46,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 1.243506015,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 1.378856619,
        "budget_committed_upper_usd": 1.378856619,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 74,
        "median": 40.130112,
        "p90": 160.974282,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 74,
        "median": 40.130112,
        "p90": 160.974282,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 74,
        "median": 0.0,
        "p90": 28,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 9,
        "median": 14.23762,
        "p90": 55.397794,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 74,
        "median": 39.868176,
        "p90": 119.225277,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "execution_segment": "initial_concurrency16",
      "scheduled_decisions": 7,
      "distinct_companies": 2,
      "outcomes": {
        "correct": 6,
        "wrong": 0,
        "cannot_verify": 1,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 7
      },
      "eligible_membership_decisions": 7,
      "full_visible_answers_reviewed": 7,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 7
      },
      "recorded_paid_http_attempts": 13,
      "recorded_api_turns": 13,
      "recorded_http_attempts_including_prior_capacity_rejections": 13,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 13,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "original_followup": 7
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 1,
          "median": 2,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 1,
          "median": 2,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 1,
          "median": 1,
          "p90": 1,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.037609651,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.041370615,
        "budget_committed_upper_usd": 0.041370615,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 7,
        "median": 22.623293,
        "p90": 47.68745,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 7,
        "median": 22.623293,
        "p90": 47.68745,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 7,
        "median": 0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 7,
        "median": 22.619500000000002,
        "p90": 47.681845,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "execution_segment": "tier_completion_concurrency16",
      "scheduled_decisions": 879,
      "distinct_companies": 92,
      "outcomes": {
        "correct": 453,
        "wrong": 31,
        "cannot_verify": 393,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 2,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 877,
        "truncated_output": 2
      },
      "eligible_membership_decisions": 877,
      "full_visible_answers_reviewed": 878,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 1,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 699,
        "query_identity_error": 2,
        "unclear": 177
      },
      "recorded_paid_http_attempts": 1149,
      "recorded_api_turns": 1149,
      "recorded_http_attempts_including_prior_capacity_rejections": 1356,
      "documented_uncharged_capacity_rejections": 207,
      "generated_response_attempts": 1149,
      "decisions_with_prior_capacity_rejection": 5,
      "completed_after_prior_capacity_rejection": 5,
      "execution_phases": {
        "tier_completion": 879
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 529,
          "median": 5,
          "p90": 6,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 529,
          "median": 4,
          "p90": 6,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 529,
          "median": 1,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 16.460215005,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 18.678236538,
        "budget_committed_upper_usd": 18.678236538,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 879,
        "median": 18.615915,
        "p90": 51.781768,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 877,
        "median": 18.576796,
        "p90": 51.781768,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 879,
        "median": 0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 5,
        "median": 283.79864200000003,
        "p90": 357.0459649999999,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 879,
        "median": 18.506695,
        "p90": 51.77858,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    }
  ],
  "by_model_arm_execution_segment": [
    {
      "model": "gpt-6-luna",
      "arm": "D",
      "execution_segment": "capacity_amendment_concurrency8",
      "scheduled_decisions": 9,
      "distinct_companies": 9,
      "outcomes": {
        "correct": 5,
        "wrong": 0,
        "cannot_verify": 4,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 9
      },
      "eligible_membership_decisions": 9,
      "full_visible_answers_reviewed": 9,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 9
      },
      "recorded_paid_http_attempts": 9,
      "recorded_api_turns": 9,
      "recorded_http_attempts_including_prior_capacity_rejections": 15,
      "documented_uncharged_capacity_rejections": 6,
      "generated_response_attempts": 9,
      "decisions_with_prior_capacity_rejection": 2,
      "completed_after_prior_capacity_rejection": 2,
      "execution_phases": {
        "amendment": 9
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.00137305,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.001510355,
        "budget_committed_upper_usd": 0.001510355,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 9,
        "median": 18.69913,
        "p90": 45.676921,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 9,
        "median": 18.69913,
        "p90": 45.676921,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 9,
        "median": 0,
        "p90": 12,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 2,
        "median": 14.20524,
        "p90": 14.23762,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 9,
        "median": 18.697637,
        "p90": 40.253237,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "D",
      "execution_segment": "tier_completion_concurrency16",
      "scheduled_decisions": 87,
      "distinct_companies": 87,
      "outcomes": {
        "correct": 31,
        "wrong": 0,
        "cannot_verify": 56,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 87
      },
      "eligible_membership_decisions": 87,
      "full_visible_answers_reviewed": 87,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 87
      },
      "recorded_paid_http_attempts": 87,
      "recorded_api_turns": 87,
      "recorded_http_attempts_including_prior_capacity_rejections": 87,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 87,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "tier_completion": 87
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.028861225,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.031747347,
        "budget_committed_upper_usd": 0.031747347,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 87,
        "median": 5.774701,
        "p90": 8.500296,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 87,
        "median": 5.774701,
        "p90": 8.500296,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 87,
        "median": 0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 87,
        "median": 5.770165,
        "p90": 8.495833,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "T",
      "execution_segment": "capacity_amendment_concurrency8",
      "scheduled_decisions": 7,
      "distinct_companies": 7,
      "outcomes": {
        "correct": 6,
        "wrong": 0,
        "cannot_verify": 1,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 7
      },
      "eligible_membership_decisions": 7,
      "full_visible_answers_reviewed": 7,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 7
      },
      "recorded_paid_http_attempts": 18,
      "recorded_api_turns": 18,
      "recorded_http_attempts_including_prior_capacity_rejections": 25,
      "documented_uncharged_capacity_rejections": 7,
      "generated_response_attempts": 18,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 7
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.001856975,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.002042672,
        "budget_committed_upper_usd": 0.002042672,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 7,
        "median": 37.047108,
        "p90": 85.106741,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 7,
        "median": 37.047108,
        "p90": 85.106741,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 7,
        "median": 0,
        "p90": 16,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 7,
        "median": 37.039077,
        "p90": 69.075153,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "T",
      "execution_segment": "initial_concurrency16",
      "scheduled_decisions": 2,
      "distinct_companies": 2,
      "outcomes": {
        "correct": 2,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 2
      },
      "eligible_membership_decisions": 2,
      "full_visible_answers_reviewed": 2,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 2
      },
      "recorded_paid_http_attempts": 5,
      "recorded_api_turns": 5,
      "recorded_http_attempts_including_prior_capacity_rejections": 5,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 5,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "original_followup": 2
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.000516001,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.0005676,
        "budget_committed_upper_usd": 0.0005676,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 2,
        "median": 35.1553715,
        "p90": 47.68745,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 2,
        "median": 35.1553715,
        "p90": 47.68745,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 2,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 2,
        "median": 35.1506725,
        "p90": 47.681845,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "T",
      "execution_segment": "tier_completion_concurrency16",
      "scheduled_decisions": 87,
      "distinct_companies": 87,
      "outcomes": {
        "correct": 83,
        "wrong": 0,
        "cannot_verify": 4,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 87
      },
      "eligible_membership_decisions": 87,
      "full_visible_answers_reviewed": 87,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 85,
        "query_identity_error": 2
      },
      "recorded_paid_http_attempts": 223,
      "recorded_api_turns": 223,
      "recorded_http_attempts_including_prior_capacity_rejections": 223,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 223,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "tier_completion": 87
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.04998344,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.054981798,
        "budget_committed_upper_usd": 0.054981798,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 87,
        "median": 7.210115,
        "p90": 10.113818,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 87,
        "median": 7.210115,
        "p90": 10.113818,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 87,
        "median": 0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 87,
        "median": 7.199536,
        "p90": 10.100954999999999,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-high",
      "execution_segment": "capacity_amendment_concurrency8",
      "scheduled_decisions": 7,
      "distinct_companies": 7,
      "outcomes": {
        "correct": 3,
        "wrong": 0,
        "cannot_verify": 4,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 7
      },
      "eligible_membership_decisions": 7,
      "full_visible_answers_reviewed": 7,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 7
      },
      "recorded_paid_http_attempts": 7,
      "recorded_api_turns": 7,
      "recorded_http_attempts_including_prior_capacity_rejections": 38,
      "documented_uncharged_capacity_rejections": 31,
      "generated_response_attempts": 7,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 7
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 7,
          "median": 3,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 7,
          "median": 3,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 7,
          "median": 1,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.097294465,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.107023913,
        "budget_committed_upper_usd": 0.107023913,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 7,
        "median": 160.974282,
        "p90": 367.933403,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 7,
        "median": 160.974282,
        "p90": 367.933403,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 7,
        "median": 60,
        "p90": 150,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 13.369162,
        "p90": 13.369162,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 7,
        "median": 111.950176,
        "p90": 217.878313,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-high",
      "execution_segment": "tier_completion_concurrency16",
      "scheduled_decisions": 89,
      "distinct_companies": 89,
      "outcomes": {
        "correct": 18,
        "wrong": 14,
        "cannot_verify": 57,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 89
      },
      "eligible_membership_decisions": 89,
      "full_visible_answers_reviewed": 89,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 75,
        "unclear": 14
      },
      "recorded_paid_http_attempts": 89,
      "recorded_api_turns": 89,
      "recorded_http_attempts_including_prior_capacity_rejections": 113,
      "documented_uncharged_capacity_rejections": 24,
      "generated_response_attempts": 89,
      "decisions_with_prior_capacity_rejection": 2,
      "completed_after_prior_capacity_rejection": 2,
      "execution_phases": {
        "tier_completion": 89
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 89,
          "median": 5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 89,
          "median": 4,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 89,
          "median": 2,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 1.54569334,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 1.942262684,
        "budget_committed_upper_usd": 1.942262684,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 89,
        "median": 21.3779,
        "p90": 26.963281,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 89,
        "median": 21.3779,
        "p90": 26.963281,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 89,
        "median": 0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 2,
        "median": 260.529554,
        "p90": 308.388281,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 89,
        "median": 21.371327,
        "p90": 26.959075,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-low",
      "execution_segment": "capacity_amendment_concurrency8",
      "scheduled_decisions": 8,
      "distinct_companies": 8,
      "outcomes": {
        "correct": 3,
        "wrong": 1,
        "cannot_verify": 4,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 8
      },
      "eligible_membership_decisions": 8,
      "full_visible_answers_reviewed": 8,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 3,
        "unclear": 5
      },
      "recorded_paid_http_attempts": 8,
      "recorded_api_turns": 8,
      "recorded_http_attempts_including_prior_capacity_rejections": 27,
      "documented_uncharged_capacity_rejections": 19,
      "generated_response_attempts": 8,
      "decisions_with_prior_capacity_rejection": 2,
      "completed_after_prior_capacity_rejection": 2,
      "execution_phases": {
        "amendment": 8
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 8,
          "median": 4.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 8,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 8,
          "median": 1.5,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.11963975,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.142603726,
        "budget_committed_upper_usd": 0.142603726,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 8,
        "median": 84.38451649999999,
        "p90": 366.537828,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 8,
        "median": 84.38451649999999,
        "p90": 366.537828,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 8,
        "median": 4.0,
        "p90": 208,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 2,
        "median": 47.4434715,
        "p90": 54.000262,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 8,
        "median": 83.37927099999999,
        "p90": 158.490072,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-low",
      "execution_segment": "tier_completion_concurrency16",
      "scheduled_decisions": 88,
      "distinct_companies": 88,
      "outcomes": {
        "correct": 19,
        "wrong": 7,
        "cannot_verify": 62,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 88
      },
      "eligible_membership_decisions": 88,
      "full_visible_answers_reviewed": 88,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 19,
        "unclear": 69
      },
      "recorded_paid_http_attempts": 88,
      "recorded_api_turns": 88,
      "recorded_http_attempts_including_prior_capacity_rejections": 88,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 88,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "tier_completion": 88
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 88,
          "median": 5.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 88,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 88,
          "median": 2.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 1.534946165,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 1.897440788,
        "budget_committed_upper_usd": 1.897440788,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 88,
        "median": 19.022237,
        "p90": 26.932827,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 88,
        "median": 19.022237,
        "p90": 26.932827,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 88,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 88,
        "median": 19.0161675,
        "p90": 26.927498,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W8-low",
      "execution_segment": "capacity_amendment_concurrency8",
      "scheduled_decisions": 6,
      "distinct_companies": 6,
      "outcomes": {
        "correct": 4,
        "wrong": 0,
        "cannot_verify": 2,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 6
      },
      "eligible_membership_decisions": 6,
      "full_visible_answers_reviewed": 6,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 6
      },
      "recorded_paid_http_attempts": 6,
      "recorded_api_turns": 6,
      "recorded_http_attempts_including_prior_capacity_rejections": 29,
      "documented_uncharged_capacity_rejections": 23,
      "generated_response_attempts": 6,
      "decisions_with_prior_capacity_rejection": 2,
      "completed_after_prior_capacity_rejection": 2,
      "execution_phases": {
        "amendment": 6
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 6,
          "median": 2.5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 6,
          "median": 2.5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 6,
          "median": 2.0,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.116922375,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.128614613,
        "budget_committed_upper_usd": 0.128614613,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 6,
        "median": 145.767932,
        "p90": 723.731584,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 6,
        "median": 145.767932,
        "p90": 723.731584,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 6,
        "median": 20.0,
        "p90": 240,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 2,
        "median": 36.08842,
        "p90": 55.397794,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 6,
        "median": 105.47486800000001,
        "p90": 483.63006599999994,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W8-low",
      "execution_segment": "tier_completion_concurrency16",
      "scheduled_decisions": 90,
      "distinct_companies": 90,
      "outcomes": {
        "correct": 31,
        "wrong": 10,
        "cannot_verify": 47,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 2,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 88,
        "truncated_output": 2
      },
      "eligible_membership_decisions": 88,
      "full_visible_answers_reviewed": 89,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 1,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 79,
        "unclear": 10
      },
      "recorded_paid_http_attempts": 90,
      "recorded_api_turns": 90,
      "recorded_http_attempts_including_prior_capacity_rejections": 126,
      "documented_uncharged_capacity_rejections": 36,
      "generated_response_attempts": 90,
      "decisions_with_prior_capacity_rejection": 3,
      "completed_after_prior_capacity_rejection": 3,
      "execution_phases": {
        "tier_completion": 90
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 90,
          "median": 5.0,
          "p90": 9,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 90,
          "median": 5.0,
          "p90": 8,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 90,
          "median": 2.0,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 2.093631335,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 2.357994471,
        "budget_committed_upper_usd": 2.357994471,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 90,
        "median": 20.718393,
        "p90": 36.846413,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 88,
        "median": 20.517339999999997,
        "p90": 36.567452,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 90,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 3,
        "median": 283.79864200000003,
        "p90": 357.0459649999999,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 90,
        "median": 20.7145385,
        "p90": 36.840031,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "D",
      "execution_segment": "capacity_amendment_concurrency8",
      "scheduled_decisions": 6,
      "distinct_companies": 6,
      "outcomes": {
        "correct": 5,
        "wrong": 0,
        "cannot_verify": 1,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 6
      },
      "eligible_membership_decisions": 6,
      "full_visible_answers_reviewed": 6,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 6
      },
      "recorded_paid_http_attempts": 6,
      "recorded_api_turns": 6,
      "recorded_http_attempts_including_prior_capacity_rejections": 7,
      "documented_uncharged_capacity_rejections": 1,
      "generated_response_attempts": 6,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 6
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.016723,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.0183953,
        "budget_committed_upper_usd": 0.0183953,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 6,
        "median": 15.4108225,
        "p90": 36.662828,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 6,
        "median": 15.4108225,
        "p90": 36.662828,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 6,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 6,
        "median": 15.4073365,
        "p90": 34.657631,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "D",
      "execution_segment": "initial_concurrency16",
      "scheduled_decisions": 2,
      "distinct_companies": 2,
      "outcomes": {
        "correct": 1,
        "wrong": 0,
        "cannot_verify": 1,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 2
      },
      "eligible_membership_decisions": 2,
      "full_visible_answers_reviewed": 2,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 2
      },
      "recorded_paid_http_attempts": 2,
      "recorded_api_turns": 2,
      "recorded_http_attempts_including_prior_capacity_rejections": 2,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 2,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "original_followup": 2
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.005104,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.0056144,
        "budget_committed_upper_usd": 0.0056144,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 2,
        "median": 13.310568,
        "p90": 14.619144,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 2,
        "median": 13.310568,
        "p90": 14.619144,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 2,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 2,
        "median": 13.309625,
        "p90": 14.617811,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "D",
      "execution_segment": "tier_completion_concurrency16",
      "scheduled_decisions": 88,
      "distinct_companies": 88,
      "outcomes": {
        "correct": 59,
        "wrong": 0,
        "cannot_verify": 29,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 88
      },
      "eligible_membership_decisions": 88,
      "full_visible_answers_reviewed": 88,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 88
      },
      "recorded_paid_http_attempts": 88,
      "recorded_api_turns": 88,
      "recorded_http_attempts_including_prior_capacity_rejections": 102,
      "documented_uncharged_capacity_rejections": 14,
      "generated_response_attempts": 88,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "tier_completion": 88
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.25524565,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.280770215,
        "budget_committed_upper_usd": 0.280770215,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 88,
        "median": 13.288379500000001,
        "p90": 17.869571,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 88,
        "median": 13.288379500000001,
        "p90": 17.869571,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 88,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 88,
        "median": 13.235485,
        "p90": 17.329747,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "T",
      "execution_segment": "capacity_amendment_concurrency8",
      "scheduled_decisions": 6,
      "distinct_companies": 6,
      "outcomes": {
        "correct": 6,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 6
      },
      "eligible_membership_decisions": 6,
      "full_visible_answers_reviewed": 6,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 6
      },
      "recorded_paid_http_attempts": 14,
      "recorded_api_turns": 14,
      "recorded_http_attempts_including_prior_capacity_rejections": 17,
      "documented_uncharged_capacity_rejections": 3,
      "generated_response_attempts": 14,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 6
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.02869125,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.031560375,
        "budget_committed_upper_usd": 0.031560375,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 6,
        "median": 24.105233,
        "p90": 35.382074,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 6,
        "median": 24.105233,
        "p90": 35.382074,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 6,
        "median": 0.0,
        "p90": 8,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 6,
        "median": 24.0978915,
        "p90": 30.916621,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "T",
      "execution_segment": "initial_concurrency16",
      "scheduled_decisions": 2,
      "distinct_companies": 2,
      "outcomes": {
        "correct": 2,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 2
      },
      "eligible_membership_decisions": 2,
      "full_visible_answers_reviewed": 2,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 2
      },
      "recorded_paid_http_attempts": 5,
      "recorded_api_turns": 5,
      "recorded_http_attempts_including_prior_capacity_rejections": 5,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 5,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "original_followup": 2
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.00995625,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.010951875,
        "budget_committed_upper_usd": 0.010951875,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 2,
        "median": 25.574428500000003,
        "p90": 33.656553,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 2,
        "median": 25.574428500000003,
        "p90": 33.656553,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 2,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 2,
        "median": 25.569697,
        "p90": 33.649864,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "T",
      "execution_segment": "tier_completion_concurrency16",
      "scheduled_decisions": 88,
      "distinct_companies": 88,
      "outcomes": {
        "correct": 87,
        "wrong": 0,
        "cannot_verify": 1,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 88
      },
      "eligible_membership_decisions": 88,
      "full_visible_answers_reviewed": 88,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 88
      },
      "recorded_paid_http_attempts": 222,
      "recorded_api_turns": 222,
      "recorded_http_attempts_including_prior_capacity_rejections": 239,
      "documented_uncharged_capacity_rejections": 17,
      "generated_response_attempts": 222,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "tier_completion": 88
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.4480528,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.49285808,
        "budget_committed_upper_usd": 0.49285808,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 88,
        "median": 17.0536235,
        "p90": 31.825721,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 88,
        "median": 17.0536235,
        "p90": 31.825721,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 88,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 88,
        "median": 16.199115499999998,
        "p90": 28.560328,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-high",
      "execution_segment": "capacity_amendment_concurrency8",
      "scheduled_decisions": 9,
      "distinct_companies": 9,
      "outcomes": {
        "correct": 5,
        "wrong": 0,
        "cannot_verify": 4,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 9
      },
      "eligible_membership_decisions": 9,
      "full_visible_answers_reviewed": 9,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 9
      },
      "recorded_paid_http_attempts": 9,
      "recorded_api_turns": 9,
      "recorded_http_attempts_including_prior_capacity_rejections": 16,
      "documented_uncharged_capacity_rejections": 7,
      "generated_response_attempts": 9,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 9
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 9,
          "median": 3,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 9,
          "median": 3,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 9,
          "median": 1,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.29295495,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.322250445,
        "budget_committed_upper_usd": 0.322250445,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 9,
        "median": 44.448006,
        "p90": 93.83099,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 9,
        "median": 44.448006,
        "p90": 93.83099,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 9,
        "median": 0,
        "p90": 14,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 11.049299,
        "p90": 11.049299,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 9,
        "median": 44.445991,
        "p90": 79.807165,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-high",
      "execution_segment": "tier_completion_concurrency16",
      "scheduled_decisions": 87,
      "distinct_companies": 87,
      "outcomes": {
        "correct": 41,
        "wrong": 0,
        "cannot_verify": 46,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 87
      },
      "eligible_membership_decisions": 87,
      "full_visible_answers_reviewed": 87,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 87
      },
      "recorded_paid_http_attempts": 87,
      "recorded_api_turns": 87,
      "recorded_http_attempts_including_prior_capacity_rejections": 124,
      "documented_uncharged_capacity_rejections": 37,
      "generated_response_attempts": 87,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "tier_completion": 87
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 87,
          "median": 5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 87,
          "median": 4,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 87,
          "median": 1,
          "p90": 1,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 3.26772945,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 3.616502395,
        "budget_committed_upper_usd": 3.616502395,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 87,
        "median": 40.907887,
        "p90": 80.362308,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 87,
        "median": 40.907887,
        "p90": 80.362308,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 87,
        "median": 0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 87,
        "median": 40.742046,
        "p90": 73.810778,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-low",
      "execution_segment": "capacity_amendment_concurrency8",
      "scheduled_decisions": 8,
      "distinct_companies": 8,
      "outcomes": {
        "correct": 4,
        "wrong": 0,
        "cannot_verify": 4,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 8
      },
      "eligible_membership_decisions": 8,
      "full_visible_answers_reviewed": 8,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 4,
        "unclear": 4
      },
      "recorded_paid_http_attempts": 8,
      "recorded_api_turns": 8,
      "recorded_http_attempts_including_prior_capacity_rejections": 10,
      "documented_uncharged_capacity_rejections": 2,
      "generated_response_attempts": 8,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 8
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 8,
          "median": 4.5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 8,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 8,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.26836495,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.295201445,
        "budget_committed_upper_usd": 0.295201445,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 8,
        "median": 40.7591735,
        "p90": 83.701051,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 8,
        "median": 40.7591735,
        "p90": 83.701051,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 8,
        "median": 0.0,
        "p90": 6,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 8,
        "median": 40.755736999999996,
        "p90": 77.68619000000001,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-low",
      "execution_segment": "initial_concurrency16",
      "scheduled_decisions": 1,
      "distinct_companies": 1,
      "outcomes": {
        "correct": 1,
        "wrong": 0,
        "cannot_verify": 0,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 1
      },
      "eligible_membership_decisions": 1,
      "full_visible_answers_reviewed": 1,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 1
      },
      "recorded_paid_http_attempts": 1,
      "recorded_api_turns": 1,
      "recorded_http_attempts_including_prior_capacity_rejections": 1,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 1,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "original_followup": 1
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 1,
          "median": 2,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 1,
          "median": 2,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 1,
          "median": 1,
          "p90": 1,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.0220334,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.02423674,
        "budget_committed_upper_usd": 0.02423674,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 1,
        "median": 32.265894,
        "p90": 32.265894,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 1,
        "median": 32.265894,
        "p90": 32.265894,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 1,
        "median": 0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 1,
        "median": 32.265246,
        "p90": 32.265246,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-low",
      "execution_segment": "tier_completion_concurrency16",
      "scheduled_decisions": 87,
      "distinct_companies": 87,
      "outcomes": {
        "correct": 34,
        "wrong": 0,
        "cannot_verify": 53,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 87
      },
      "eligible_membership_decisions": 87,
      "full_visible_answers_reviewed": 87,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 34,
        "unclear": 53
      },
      "recorded_paid_http_attempts": 87,
      "recorded_api_turns": 87,
      "recorded_http_attempts_including_prior_capacity_rejections": 120,
      "documented_uncharged_capacity_rejections": 33,
      "generated_response_attempts": 87,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "tier_completion": 87
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 87,
          "median": 5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 87,
          "median": 4,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 87,
          "median": 1,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 3.36525025,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 3.745775275,
        "budget_committed_upper_usd": 3.745775275,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 87,
        "median": 42.796156,
        "p90": 76.2564,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 87,
        "median": 42.796156,
        "p90": 76.2564,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 87,
        "median": 0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 87,
        "median": 42.620025,
        "p90": 73.118786,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W8-low",
      "execution_segment": "capacity_amendment_concurrency8",
      "scheduled_decisions": 8,
      "distinct_companies": 8,
      "outcomes": {
        "correct": 5,
        "wrong": 0,
        "cannot_verify": 3,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 8
      },
      "eligible_membership_decisions": 8,
      "full_visible_answers_reviewed": 8,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 8
      },
      "recorded_paid_http_attempts": 8,
      "recorded_api_turns": 8,
      "recorded_http_attempts_including_prior_capacity_rejections": 11,
      "documented_uncharged_capacity_rejections": 3,
      "generated_response_attempts": 8,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 8
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 8,
          "median": 4.5,
          "p90": 6,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 8,
          "median": 4.5,
          "p90": 6,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 8,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.29968525,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.329653775,
        "budget_committed_upper_usd": 0.329653775,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 8,
        "median": 50.2679575,
        "p90": 73.634111,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 8,
        "median": 50.2679575,
        "p90": 73.634111,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 8,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 11.064242,
        "p90": 11.064242,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 8,
        "median": 50.2646945,
        "p90": 73.631286,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W8-low",
      "execution_segment": "tier_completion_concurrency16",
      "scheduled_decisions": 88,
      "distinct_companies": 88,
      "outcomes": {
        "correct": 50,
        "wrong": 0,
        "cannot_verify": 38,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 88
      },
      "eligible_membership_decisions": 88,
      "full_visible_answers_reviewed": 88,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 57,
        "unclear": 31
      },
      "recorded_paid_http_attempts": 88,
      "recorded_api_turns": 88,
      "recorded_http_attempts_including_prior_capacity_rejections": 134,
      "documented_uncharged_capacity_rejections": 46,
      "generated_response_attempts": 88,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "tier_completion": 88
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 88,
          "median": 5.0,
          "p90": 6,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 88,
          "median": 5.0,
          "p90": 6,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 88,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 3.87082135,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 4.257903485,
        "budget_committed_upper_usd": 4.257903485,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 88,
        "median": 42.9055205,
        "p90": 110.567648,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 88,
        "median": 42.9055205,
        "p90": 110.567648,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 88,
        "median": 0.0,
        "p90": 6,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 88,
        "median": 42.7976095,
        "p90": 95.114824,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    }
  ],
  "by_model_arm_requested_service_tier": [
    {
      "model": "gpt-6-luna",
      "arm": "D",
      "requested_service_tier": "default",
      "scheduled_decisions": 87,
      "distinct_companies": 87,
      "outcomes": {
        "correct": 31,
        "wrong": 0,
        "cannot_verify": 56,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 87
      },
      "eligible_membership_decisions": 87,
      "full_visible_answers_reviewed": 87,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 87
      },
      "recorded_paid_http_attempts": 87,
      "recorded_api_turns": 87,
      "recorded_http_attempts_including_prior_capacity_rejections": 87,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 87,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "tier_completion": 87
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.028861225,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.031747347,
        "budget_committed_upper_usd": 0.031747347,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 87,
        "median": 5.774701,
        "p90": 8.500296,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 87,
        "median": 5.774701,
        "p90": 8.500296,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 87,
        "median": 0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 87,
        "median": 5.770165,
        "p90": 8.495833,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "D",
      "requested_service_tier": "flex",
      "scheduled_decisions": 9,
      "distinct_companies": 9,
      "outcomes": {
        "correct": 5,
        "wrong": 0,
        "cannot_verify": 4,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 9
      },
      "eligible_membership_decisions": 9,
      "full_visible_answers_reviewed": 9,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 9
      },
      "recorded_paid_http_attempts": 9,
      "recorded_api_turns": 9,
      "recorded_http_attempts_including_prior_capacity_rejections": 15,
      "documented_uncharged_capacity_rejections": 6,
      "generated_response_attempts": 9,
      "decisions_with_prior_capacity_rejection": 2,
      "completed_after_prior_capacity_rejection": 2,
      "execution_phases": {
        "amendment": 9
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.00137305,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.001510355,
        "budget_committed_upper_usd": 0.001510355,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 9,
        "median": 18.69913,
        "p90": 45.676921,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 9,
        "median": 18.69913,
        "p90": 45.676921,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 9,
        "median": 0,
        "p90": 12,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 2,
        "median": 14.20524,
        "p90": 14.23762,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 9,
        "median": 18.697637,
        "p90": 40.253237,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "T",
      "requested_service_tier": "default",
      "scheduled_decisions": 87,
      "distinct_companies": 87,
      "outcomes": {
        "correct": 83,
        "wrong": 0,
        "cannot_verify": 4,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 87
      },
      "eligible_membership_decisions": 87,
      "full_visible_answers_reviewed": 87,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 85,
        "query_identity_error": 2
      },
      "recorded_paid_http_attempts": 223,
      "recorded_api_turns": 223,
      "recorded_http_attempts_including_prior_capacity_rejections": 223,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 223,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "tier_completion": 87
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.04998344,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.054981798,
        "budget_committed_upper_usd": 0.054981798,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 87,
        "median": 7.210115,
        "p90": 10.113818,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 87,
        "median": 7.210115,
        "p90": 10.113818,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 87,
        "median": 0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 87,
        "median": 7.199536,
        "p90": 10.100954999999999,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "T",
      "requested_service_tier": "flex",
      "scheduled_decisions": 9,
      "distinct_companies": 9,
      "outcomes": {
        "correct": 8,
        "wrong": 0,
        "cannot_verify": 1,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 9
      },
      "eligible_membership_decisions": 9,
      "full_visible_answers_reviewed": 9,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 9
      },
      "recorded_paid_http_attempts": 23,
      "recorded_api_turns": 23,
      "recorded_http_attempts_including_prior_capacity_rejections": 30,
      "documented_uncharged_capacity_rejections": 7,
      "generated_response_attempts": 23,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 7,
        "original_followup": 2
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.002372976,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.002610272,
        "budget_committed_upper_usd": 0.002610272,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 9,
        "median": 37.047108,
        "p90": 85.106741,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 9,
        "median": 37.047108,
        "p90": 85.106741,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 9,
        "median": 0,
        "p90": 16,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 9,
        "median": 37.039077,
        "p90": 69.075153,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-high",
      "requested_service_tier": "default",
      "scheduled_decisions": 89,
      "distinct_companies": 89,
      "outcomes": {
        "correct": 18,
        "wrong": 14,
        "cannot_verify": 57,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 89
      },
      "eligible_membership_decisions": 89,
      "full_visible_answers_reviewed": 89,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 75,
        "unclear": 14
      },
      "recorded_paid_http_attempts": 89,
      "recorded_api_turns": 89,
      "recorded_http_attempts_including_prior_capacity_rejections": 113,
      "documented_uncharged_capacity_rejections": 24,
      "generated_response_attempts": 89,
      "decisions_with_prior_capacity_rejection": 2,
      "completed_after_prior_capacity_rejection": 2,
      "execution_phases": {
        "tier_completion": 89
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 89,
          "median": 5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 89,
          "median": 4,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 89,
          "median": 2,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 1.54569334,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 1.942262684,
        "budget_committed_upper_usd": 1.942262684,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 89,
        "median": 21.3779,
        "p90": 26.963281,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 89,
        "median": 21.3779,
        "p90": 26.963281,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 89,
        "median": 0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 2,
        "median": 260.529554,
        "p90": 308.388281,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 89,
        "median": 21.371327,
        "p90": 26.959075,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-high",
      "requested_service_tier": "flex",
      "scheduled_decisions": 7,
      "distinct_companies": 7,
      "outcomes": {
        "correct": 3,
        "wrong": 0,
        "cannot_verify": 4,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 7
      },
      "eligible_membership_decisions": 7,
      "full_visible_answers_reviewed": 7,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 7
      },
      "recorded_paid_http_attempts": 7,
      "recorded_api_turns": 7,
      "recorded_http_attempts_including_prior_capacity_rejections": 38,
      "documented_uncharged_capacity_rejections": 31,
      "generated_response_attempts": 7,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 7
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 7,
          "median": 3,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 7,
          "median": 3,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 7,
          "median": 1,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.097294465,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.107023913,
        "budget_committed_upper_usd": 0.107023913,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 7,
        "median": 160.974282,
        "p90": 367.933403,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 7,
        "median": 160.974282,
        "p90": 367.933403,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 7,
        "median": 60,
        "p90": 150,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 13.369162,
        "p90": 13.369162,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 7,
        "median": 111.950176,
        "p90": 217.878313,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-low",
      "requested_service_tier": "default",
      "scheduled_decisions": 88,
      "distinct_companies": 88,
      "outcomes": {
        "correct": 19,
        "wrong": 7,
        "cannot_verify": 62,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 88
      },
      "eligible_membership_decisions": 88,
      "full_visible_answers_reviewed": 88,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 19,
        "unclear": 69
      },
      "recorded_paid_http_attempts": 88,
      "recorded_api_turns": 88,
      "recorded_http_attempts_including_prior_capacity_rejections": 88,
      "documented_uncharged_capacity_rejections": 0,
      "generated_response_attempts": 88,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "tier_completion": 88
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 88,
          "median": 5.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 88,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 88,
          "median": 2.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 1.534946165,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 1.897440788,
        "budget_committed_upper_usd": 1.897440788,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 88,
        "median": 19.022237,
        "p90": 26.932827,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 88,
        "median": 19.022237,
        "p90": 26.932827,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 88,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 88,
        "median": 19.0161675,
        "p90": 26.927498,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W4-low",
      "requested_service_tier": "flex",
      "scheduled_decisions": 8,
      "distinct_companies": 8,
      "outcomes": {
        "correct": 3,
        "wrong": 1,
        "cannot_verify": 4,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 8
      },
      "eligible_membership_decisions": 8,
      "full_visible_answers_reviewed": 8,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 3,
        "unclear": 5
      },
      "recorded_paid_http_attempts": 8,
      "recorded_api_turns": 8,
      "recorded_http_attempts_including_prior_capacity_rejections": 27,
      "documented_uncharged_capacity_rejections": 19,
      "generated_response_attempts": 8,
      "decisions_with_prior_capacity_rejection": 2,
      "completed_after_prior_capacity_rejection": 2,
      "execution_phases": {
        "amendment": 8
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 8,
          "median": 4.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 8,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 8,
          "median": 1.5,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.11963975,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.142603726,
        "budget_committed_upper_usd": 0.142603726,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 8,
        "median": 84.38451649999999,
        "p90": 366.537828,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 8,
        "median": 84.38451649999999,
        "p90": 366.537828,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 8,
        "median": 4.0,
        "p90": 208,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 2,
        "median": 47.4434715,
        "p90": 54.000262,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 8,
        "median": 83.37927099999999,
        "p90": 158.490072,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W8-low",
      "requested_service_tier": "default",
      "scheduled_decisions": 90,
      "distinct_companies": 90,
      "outcomes": {
        "correct": 31,
        "wrong": 10,
        "cannot_verify": 47,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 2,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 88,
        "truncated_output": 2
      },
      "eligible_membership_decisions": 88,
      "full_visible_answers_reviewed": 89,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 1,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 79,
        "unclear": 10
      },
      "recorded_paid_http_attempts": 90,
      "recorded_api_turns": 90,
      "recorded_http_attempts_including_prior_capacity_rejections": 126,
      "documented_uncharged_capacity_rejections": 36,
      "generated_response_attempts": 90,
      "decisions_with_prior_capacity_rejection": 3,
      "completed_after_prior_capacity_rejection": 3,
      "execution_phases": {
        "tier_completion": 90
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 90,
          "median": 5.0,
          "p90": 9,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 90,
          "median": 5.0,
          "p90": 8,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 90,
          "median": 2.0,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 2.093631335,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 2.357994471,
        "budget_committed_upper_usd": 2.357994471,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 90,
        "median": 20.718393,
        "p90": 36.846413,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 88,
        "median": 20.517339999999997,
        "p90": 36.567452,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 90,
        "median": 0.0,
        "p90": 0,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 3,
        "median": 283.79864200000003,
        "p90": 357.0459649999999,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 90,
        "median": 20.7145385,
        "p90": 36.840031,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6-luna",
      "arm": "W8-low",
      "requested_service_tier": "flex",
      "scheduled_decisions": 6,
      "distinct_companies": 6,
      "outcomes": {
        "correct": 4,
        "wrong": 0,
        "cannot_verify": 2,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 6
      },
      "eligible_membership_decisions": 6,
      "full_visible_answers_reviewed": 6,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 6
      },
      "recorded_paid_http_attempts": 6,
      "recorded_api_turns": 6,
      "recorded_http_attempts_including_prior_capacity_rejections": 29,
      "documented_uncharged_capacity_rejections": 23,
      "generated_response_attempts": 6,
      "decisions_with_prior_capacity_rejection": 2,
      "completed_after_prior_capacity_rejection": 2,
      "execution_phases": {
        "amendment": 6
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 6,
          "median": 2.5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 6,
          "median": 2.5,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 6,
          "median": 2.0,
          "p90": 3,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.116922375,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.128614613,
        "budget_committed_upper_usd": 0.128614613,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 6,
        "median": 145.767932,
        "p90": 723.731584,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 6,
        "median": 145.767932,
        "p90": 723.731584,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 6,
        "median": 20.0,
        "p90": 240,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 2,
        "median": 36.08842,
        "p90": 55.397794,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 6,
        "median": 105.47486800000001,
        "p90": 483.63006599999994,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "D",
      "requested_service_tier": "flex",
      "scheduled_decisions": 96,
      "distinct_companies": 96,
      "outcomes": {
        "correct": 65,
        "wrong": 0,
        "cannot_verify": 31,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 96
      },
      "eligible_membership_decisions": 96,
      "full_visible_answers_reviewed": 96,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 96
      },
      "recorded_paid_http_attempts": 96,
      "recorded_api_turns": 96,
      "recorded_http_attempts_including_prior_capacity_rejections": 111,
      "documented_uncharged_capacity_rejections": 15,
      "generated_response_attempts": 96,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 6,
        "original_followup": 2,
        "tier_completion": 88
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.27707265,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.304779915,
        "budget_committed_upper_usd": 0.304779915,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 96,
        "median": 13.361460000000001,
        "p90": 18.007425,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 96,
        "median": 13.361460000000001,
        "p90": 18.007425,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 96,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 96,
        "median": 13.284182999999999,
        "p90": 17.716105,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "T",
      "requested_service_tier": "flex",
      "scheduled_decisions": 96,
      "distinct_companies": 96,
      "outcomes": {
        "correct": 95,
        "wrong": 0,
        "cannot_verify": 1,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 96
      },
      "eligible_membership_decisions": 96,
      "full_visible_answers_reviewed": 96,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 96
      },
      "recorded_paid_http_attempts": 241,
      "recorded_api_turns": 241,
      "recorded_http_attempts_including_prior_capacity_rejections": 261,
      "documented_uncharged_capacity_rejections": 20,
      "generated_response_attempts": 241,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 6,
        "original_followup": 2,
        "tier_completion": 88
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 0,
          "median": null,
          "p90": null,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 0.4867003,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 0.53537033,
        "budget_committed_upper_usd": 0.53537033,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 96,
        "median": 17.2316485,
        "p90": 31.872571,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 96,
        "median": 17.2316485,
        "p90": 31.872571,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 96,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 96,
        "median": 17.2239065,
        "p90": 29.481018000000002,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-high",
      "requested_service_tier": "flex",
      "scheduled_decisions": 96,
      "distinct_companies": 96,
      "outcomes": {
        "correct": 46,
        "wrong": 0,
        "cannot_verify": 50,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 96
      },
      "eligible_membership_decisions": 96,
      "full_visible_answers_reviewed": 96,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 96
      },
      "recorded_paid_http_attempts": 96,
      "recorded_api_turns": 96,
      "recorded_http_attempts_including_prior_capacity_rejections": 140,
      "documented_uncharged_capacity_rejections": 44,
      "generated_response_attempts": 96,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 9,
        "tier_completion": 87
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 96,
          "median": 5.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 96,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 96,
          "median": 1.0,
          "p90": 1,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 3.5606844,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 3.93875284,
        "budget_committed_upper_usd": 3.93875284,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 96,
        "median": 40.9654135,
        "p90": 80.362308,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 96,
        "median": 40.9654135,
        "p90": 80.362308,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 96,
        "median": 0.0,
        "p90": 6,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 11.049299,
        "p90": 11.049299,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 96,
        "median": 40.8222735,
        "p90": 73.810778,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W4-low",
      "requested_service_tier": "flex",
      "scheduled_decisions": 96,
      "distinct_companies": 96,
      "outcomes": {
        "correct": 39,
        "wrong": 0,
        "cannot_verify": 57,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 96
      },
      "eligible_membership_decisions": 96,
      "full_visible_answers_reviewed": 96,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 39,
        "unclear": 57
      },
      "recorded_paid_http_attempts": 96,
      "recorded_api_turns": 96,
      "recorded_http_attempts_including_prior_capacity_rejections": 131,
      "documented_uncharged_capacity_rejections": 35,
      "generated_response_attempts": 96,
      "decisions_with_prior_capacity_rejection": 0,
      "completed_after_prior_capacity_rejection": 0,
      "execution_phases": {
        "amendment": 8,
        "original_followup": 1,
        "tier_completion": 87
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 96,
          "median": 5.0,
          "p90": 5,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 96,
          "median": 4.0,
          "p90": 4,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 96,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 3.6556486,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 4.06521346,
        "budget_committed_upper_usd": 4.06521346,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 96,
        "median": 42.6963,
        "p90": 76.2564,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 96,
        "median": 42.6963,
        "p90": 76.2564,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 96,
        "median": 0.0,
        "p90": 2,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 0,
        "median": null,
        "p90": null,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 96,
        "median": 42.339066,
        "p90": 73.118786,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    },
    {
      "model": "gpt-6.1-sol",
      "arm": "W8-low",
      "requested_service_tier": "flex",
      "scheduled_decisions": 96,
      "distinct_companies": 96,
      "outcomes": {
        "correct": 55,
        "wrong": 0,
        "cannot_verify": 41,
        "other_answer": 0,
        "operational_failed": 0,
        "missing": 0,
        "truncated": 0,
        "contaminated": 0
      },
      "operational_statuses": {
        "complete": 96
      },
      "eligible_membership_decisions": 96,
      "full_visible_answers_reviewed": 96,
      "visible_answers_needing_review": 0,
      "operational_no_answer_reviews": 0,
      "generated_no_answer_events_needing_review": 0,
      "contamination_candidates": 0,
      "reviewed_failure_stages": {
        "not_applicable": 65,
        "unclear": 31
      },
      "recorded_paid_http_attempts": 96,
      "recorded_api_turns": 96,
      "recorded_http_attempts_including_prior_capacity_rejections": 145,
      "documented_uncharged_capacity_rejections": 49,
      "generated_response_attempts": 96,
      "decisions_with_prior_capacity_rejection": 1,
      "completed_after_prior_capacity_rejection": 1,
      "execution_phases": {
        "amendment": 8,
        "tier_completion": 88
      },
      "realized_web_actions": {
        "visible_including_unfinished": {
          "n": 96,
          "median": 5.0,
          "p90": 6,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "completed": {
          "n": 96,
          "median": 5.0,
          "p90": 6,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "search_actions": {
          "n": 96,
          "median": 1.0,
          "p90": 2,
          "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
        },
        "interpretation": "Assigned conditions specify maximum tool calls, not mandatory actual searches. Visible in-progress actions can appear beyond the requested limit; their status is preserved and not counted as a model error."
      },
      "cost": {
        "scope": "Follow-up decisions in this group only; excludes the separately reported original experiment opening amount.",
        "known_usage_based_usd": 4.1705066,
        "unknown_usage_decisions": 0,
        "observed_conservative_usd": 4.58755726,
        "budget_committed_upper_usd": 4.58755726,
        "budget_commitment_meaning": "Settled conservative costs plus retained reservations for pending or unknown-charge decisions; not observed spending.",
        "provider_invoice_observed": false
      },
      "latency_seconds_all_recorded_decisions": {
        "n": 96,
        "median": 43.8265205,
        "p90": 100.087125,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "latency_seconds_complete_eligible_decisions": {
        "n": 96,
        "median": 43.8265205,
        "p90": 100.087125,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "capacity_retry_backoff_seconds": {
        "n": 96,
        "median": 0.0,
        "p90": 6,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "prior_capacity_http_latency_seconds": {
        "n": 1,
        "median": 11.064242,
        "p90": 11.064242,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      },
      "summed_http_latency_seconds": {
        "n": 96,
        "median": 43.8215875,
        "p90": 94.058713,
        "p90_definition": "nearest rank ceil(0.9*n), without interpolation"
      }
    }
  ],
  "paired_web_conditions": [
    {
      "model": "gpt-6.1-sol",
      "baseline": "W4-low",
      "contrast": "W8-low",
      "scheduled_company_pairs": 96,
      "eligible_pairs": 96,
      "excluded_or_missing_pairs": 0,
      "transitions": {
        "cannot_verify -> cannot_verify": 39,
        "cannot_verify -> correct": 18,
        "correct -> cannot_verify": 2,
        "correct -> correct": 37
      },
      "excluded_transitions": {},
      "eligible_pairs_by_requested_tier_combination": {
        "flex -> flex": 96
      },
      "same_tier_sensitivity": [
        {
          "tier": "flex",
          "eligible_pairs": 96,
          "transitions": {
            "cannot_verify -> cannot_verify": 39,
            "cannot_verify -> correct": 18,
            "correct -> cannot_verify": 2,
            "correct -> correct": 37
          }
        },
        {
          "tier": "default",
          "eligible_pairs": 0,
          "transitions": {}
        }
      ],
      "mixed_tier_eligible_pairs": 0,
      "interpretation": "Within-company descriptive comparison; one repetition. Same-tier sensitivities are separate. Mixed-tier pairs do not establish a controlled call/context effect; no tier-equivalent behavior, significance or population error-rate claim."
    },
    {
      "model": "gpt-6.1-sol",
      "baseline": "W4-low",
      "contrast": "W4-high",
      "scheduled_company_pairs": 96,
      "eligible_pairs": 96,
      "excluded_or_missing_pairs": 0,
      "transitions": {
        "cannot_verify -> cannot_verify": 46,
        "cannot_verify -> correct": 11,
        "correct -> cannot_verify": 4,
        "correct -> correct": 35
      },
      "excluded_transitions": {},
      "eligible_pairs_by_requested_tier_combination": {
        "flex -> flex": 96
      },
      "same_tier_sensitivity": [
        {
          "tier": "flex",
          "eligible_pairs": 96,
          "transitions": {
            "cannot_verify -> cannot_verify": 46,
            "cannot_verify -> correct": 11,
            "correct -> cannot_verify": 4,
            "correct -> correct": 35
          }
        },
        {
          "tier": "default",
          "eligible_pairs": 0,
          "transitions": {}
        }
      ],
      "mixed_tier_eligible_pairs": 0,
      "interpretation": "Within-company descriptive comparison; one repetition. Same-tier sensitivities are separate. Mixed-tier pairs do not establish a controlled call/context effect; no tier-equivalent behavior, significance or population error-rate claim."
    },
    {
      "model": "gpt-6-luna",
      "baseline": "W4-low",
      "contrast": "W8-low",
      "scheduled_company_pairs": 96,
      "eligible_pairs": 94,
      "excluded_or_missing_pairs": 2,
      "transitions": {
        "cannot_verify -> cannot_verify": 40,
        "cannot_verify -> correct": 17,
        "cannot_verify -> wrong": 7,
        "correct -> cannot_verify": 6,
        "correct -> correct": 16,
        "wrong -> cannot_verify": 3,
        "wrong -> correct": 2,
        "wrong -> wrong": 3
      },
      "excluded_transitions": {
        "cannot_verify -> truncated": 2
      },
      "eligible_pairs_by_requested_tier_combination": {
        "default -> default": 85,
        "default -> flex": 1,
        "flex -> default": 3,
        "flex -> flex": 5
      },
      "same_tier_sensitivity": [
        {
          "tier": "flex",
          "eligible_pairs": 5,
          "transitions": {
            "cannot_verify -> cannot_verify": 1,
            "cannot_verify -> correct": 1,
            "correct -> cannot_verify": 1,
            "correct -> correct": 2
          }
        },
        {
          "tier": "default",
          "eligible_pairs": 85,
          "transitions": {
            "cannot_verify -> cannot_verify": 38,
            "cannot_verify -> correct": 15,
            "cannot_verify -> wrong": 7,
            "correct -> cannot_verify": 5,
            "correct -> correct": 13,
            "wrong -> cannot_verify": 3,
            "wrong -> correct": 1,
            "wrong -> wrong": 3
          }
        }
      ],
      "mixed_tier_eligible_pairs": 4,
      "interpretation": "Within-company descriptive comparison; one repetition. Same-tier sensitivities are separate. Mixed-tier pairs do not establish a controlled call/context effect; no tier-equivalent behavior, significance or population error-rate claim."
    },
    {
      "model": "gpt-6-luna",
      "baseline": "W4-low",
      "contrast": "W4-high",
      "scheduled_company_pairs": 96,
      "eligible_pairs": 96,
      "excluded_or_missing_pairs": 0,
      "transitions": {
        "cannot_verify -> cannot_verify": 47,
        "cannot_verify -> correct": 10,
        "cannot_verify -> wrong": 9,
        "correct -> cannot_verify": 11,
        "correct -> correct": 11,
        "wrong -> cannot_verify": 3,
        "wrong -> wrong": 5
      },
      "excluded_transitions": {},
      "eligible_pairs_by_requested_tier_combination": {
        "default -> default": 87,
        "default -> flex": 1,
        "flex -> default": 2,
        "flex -> flex": 6
      },
      "same_tier_sensitivity": [
        {
          "tier": "flex",
          "eligible_pairs": 6,
          "transitions": {
            "cannot_verify -> cannot_verify": 2,
            "cannot_verify -> correct": 1,
            "correct -> cannot_verify": 2,
            "correct -> correct": 1
          }
        },
        {
          "tier": "default",
          "eligible_pairs": 87,
          "transitions": {
            "cannot_verify -> cannot_verify": 44,
            "cannot_verify -> correct": 9,
            "cannot_verify -> wrong": 9,
            "correct -> cannot_verify": 9,
            "correct -> correct": 9,
            "wrong -> cannot_verify": 3,
            "wrong -> wrong": 4
          }
        }
      ],
      "mixed_tier_eligible_pairs": 3,
      "interpretation": "Within-company descriptive comparison; one repetition. Same-tier sensitivities are separate. Mixed-tier pairs do not establish a controlled call/context effect; no tier-equivalent behavior, significance or population error-rate claim."
    }
  ],
  "review_provenance": [
    {
      "file": "private/followup-review-baseline-web.json",
      "sha256": "11564b37bad0480fd6c181025cd6c62233a6d3da76bd97fbafe3e5555f203d9e"
    },
    {
      "file": "private/followup-review-expanded-web.json",
      "sha256": "f4561a4688209140821e1f5a73a5e17f26c611c312aafe3f148eb5ea73382a99"
    },
    {
      "file": "private/followup-review-docs-tool.json",
      "sha256": "5d52f5235cc60881ae274fca957ebebb0813e795cd507b3b18ffdd3cd42dc219"
    }
  ],
  "all_recorded_visible_answers_reviewed": true,
  "all_scheduled_decisions_recorded": true,
  "opening_original_conservative_usd": 20.621832436,
  "amendment_opening_including_preserved_followup_usd": 20.663203052,
  "combined_committed_upper_usd": 40.720296208,
  "limits": [
    "Original experiment and follow-up are separate; original outcomes are not pooled with new counts.",
    "The full schedule denominator is retained, including missing, failed, truncated and contaminated decisions.",
    "Machine parsing is not full-answer or source-support review. Review is internal and unblinded unless independently documented.",
    "Source dates are observed strings; older/null dates alone do not prove an incorrect verdict or source insufficiency.",
    "Self-study trace flags need review; attempted opens alone are not proof of content exposure.",
    "Latency includes all recorded function-continuation turns and local tool handling, excludes queue/reservation wait; HTTP durations are also reported separately.",
    "Amended decision latency includes automatic capacity retry/backoff within that execution. The earlier aborted-run interval and coordinator pause are retained as separate provenance and excluded from that latency; prior HTTP durations and attempts are reported separately.",
    "Concurrency changed operationally from16 initially, to8 during capacity recovery, and then16 under a separate operator protocol if present. Segment-stratified timings are descriptive; concurrency, execution order and source/cache changes prevent interpreting latency or availability differences as controlled model effects.",
    "Luna service tier changes prospectively from Flex to default for never-generated slots under the tier-completion protocol, if present; all prior generated answers remain unchanged. Same-tier paired sensitivities and mixed-tier pair counts are reported separately; tier-equivalent behavior is not assumed.",
    "Inter-phase coordinator pauses and time awaiting local dispatch are excluded from per-decision elapsed time. Provider waits inside HTTP requests and bounded retry backoff remain included.",
    "HTTP capacity rejections are operational events, not model mistakes or abstentions. The recovery amendment preserves the first generated answer and never repeats a generated answer.",
    "Cost values are estimates from recorded usage and budget reservations, not a provider invoice.",
    "The selected panel and single-provider models do not estimate population accuracy or legal clearance."
  ],
  "accounting_reconciliation": {
    "scope": "Original experiment plus follow-up; preserved prior generated decisions counted once across availability phases.",
    "original_conservative_opening_usd": 20.621832436,
    "followup_known_usage_based_usd": 17.741330671,
    "followup_observed_conservative_usd": 20.098463772,
    "combined_observed_conservative_usd": 40.720296208,
    "followup_committed_upper_including_reservations_usd": 20.098463772,
    "combined_committed_upper_including_reservations_usd": 40.720296208,
    "latest_phase": "tier_completion",
    "latest_phase_opening_including_prior_generations_usd": 22.042059667,
    "latest_phase_entry_commitments_usd": 18.678236538,
    "latest_phase_ledger_combined_upper_usd": 40.720296205,
    "logical_vs_ledger_rounding_difference_usd": 3e-09,
    "latest_phase_pending_entries": 0,
    "latest_phase_pending_reservations_usd": 0,
    "provider_invoice_observed": false
  },
  "package_status": "INTERNALLY_REVIEWED_FOLLOWUP_EVIDENCE_EXPORT",
  "analysis_reproduction_status": "PRELIMINARY_FOLLOWUP_NOT_PUBLICATION_APPROVAL",
  "internal_unblinded_review": true
}
