{
  "context": {
    "cap_tokens": 320512,
    "max_model_len": 327680,
    "targets": []
  },
  "evaluation_protocol": {
    "budget_semantics": "visible_answer_tokens plus reasoning_headroom_tokens are sent as one max_tokens cap; the endpoint does not hard-partition the channels",
    "minimum_comparable_repetitions": 3,
    "minimum_pass_rate": 1.0,
    "reasoning_headroom_tokens": 4096,
    "records_finish_reason": true,
    "records_full_visible_answer": true,
    "records_reasoning_channel_metadata": true,
    "repetitions": 3,
    "version": 3,
    "visible_answer_tokens": 2048
  },
  "failures": [],
  "identity": {
    "candidate_id": "glm-dcp1",
    "config_id": "r10-dcp1",
    "engine": "vllm",
    "gpu": "dual-pro6000",
    "model": "glm53-flash-exl3-runtime-dcp1-candidate",
    "started_at": "2026-09-23T22:35:24Z"
  },
  "intelligence": {
    "checks": [
      {
        "attempt_count": 3,
        "attempts": [
          {
            "attempt": 1,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 6144,
              "reasoning_headroom_tokens": 4096,
              "visible_answer_tokens": 2048
            },
            "content": "```diff\n--- app.py\n+++ app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "content_chars": 90,
            "content_excerpt": "```diff\n--- app.py\n+++ app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 6167.46659999626,
            "reasoning_chars": 1536,
            "reasoning_excerpt": "The user wants a unified diff that changes `timeout = 30` to `timeout = 45` in app.py, leaving `retries = 2` unchanged.\n\nA unified diff format looks like:\n\n```\n--- app.py\n+++ app.py\n@@ -1,2 +1,2 @@\n t",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "diff_shape",
                "passed": true
              },
              {
                "name": "removes_old_timeout",
                "passed": true
              },
              {
                "name": "adds_new_timeout",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 475,
              "prompt_tokens": 48,
              "prompt_tokens_details": null,
              "total_tokens": 523
            }
          },
          {
            "attempt": 2,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 6144,
              "reasoning_headroom_tokens": 4096,
              "visible_answer_tokens": 2048
            },
            "content": "```diff\n--- app.py\n+++ app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "content_chars": 90,
            "content_excerpt": "```diff\n--- app.py\n+++ app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 10068.399141004193,
            "reasoning_chars": 2576,
            "reasoning_excerpt": "The user wants a unified diff that changes `timeout = 30` to `timeout = 45` and leaves `retries = 2` unchanged.\n\nA unified diff format looks like:\n\n```\n--- app.py\n+++ app.py\n@@ -1,2 +1,2 @@\n timeout =",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "diff_shape",
                "passed": true
              },
              {
                "name": "removes_old_timeout",
                "passed": true
              },
              {
                "name": "adds_new_timeout",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 780,
              "prompt_tokens": 48,
              "prompt_tokens_details": null,
              "total_tokens": 828
            }
          },
          {
            "attempt": 3,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 6144,
              "reasoning_headroom_tokens": 4096,
              "visible_answer_tokens": 2048
            },
            "content": "```diff\n--- app.py\n+++ app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "content_chars": 90,
            "content_excerpt": "```diff\n--- app.py\n+++ app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 10157.922270002018,
            "reasoning_chars": 2681,
            "reasoning_excerpt": "The user wants a unified diff that changes `timeout = 30` to `timeout = 45` while leaving `retries = 2` unchanged.\n\nA unified diff format looks like:\n\n```\n--- app.py\n+++ app.py\n@@ -1,2 +1,2 @@\n-timeou",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "diff_shape",
                "passed": true
              },
              {
                "name": "removes_old_timeout",
                "passed": true
              },
              {
                "name": "adds_new_timeout",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 787,
              "prompt_tokens": 48,
              "prompt_tokens_details": null,
              "total_tokens": 835
            }
          }
        ],
        "id": "unified_diff_timeout_edit",
        "pass_count": 3,
        "pass_rate": 1.0,
        "required_pass_rate": 1.0,
        "status": "passed",
        "validator": "deterministic_text_checks"
      },
      {
        "attempt_count": 3,
        "attempts": [
          {
            "attempt": 1,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 6144,
              "reasoning_headroom_tokens": 4096,
              "visible_answer_tokens": 2048
            },
            "content": "The pipeline's serial latency (550 + 1800 + 650 = 3000 ms) exceeds the 2500 ms turn timeout, with the LLM being the biggest bottleneck—so stream LLM tokens into TTS (and/or switch to a faster model) so TTS starts generating while the LLM is still responding, cutting end-to-end latency under budget.",
            "content_chars": 299,
            "content_excerpt": "The pipeline's serial latency (550 + 1800 + 650 = 3000 ms) exceeds the 2500 ms turn timeout, with the LLM being the biggest bottleneck—so stream LLM tokens into TTS (and/or switch to a faster model) s",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 6174.619329998677,
            "reasoning_chars": 1562,
            "reasoning_excerpt": "The user is asking about a voice agent pipeline: STT (speech-to-text), LLM, TTS (text-to-speech). Total turn timeout is 2500 ms. Logs show STT=550 ms, LLM=1800 ms, TTS=650 ms. Sum = 550 + 1800 + 650 =",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "identifies_budget_overrun",
                "passed": true
              },
              {
                "name": "offers_latency_fix",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 471,
              "prompt_tokens": 71,
              "prompt_tokens_details": null,
              "total_tokens": 542
            }
          },
          {
            "attempt": 2,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 6144,
              "reasoning_headroom_tokens": 4096,
              "visible_answer_tokens": 2048
            },
            "content": "The serialized pipeline (550 + 1800 + 650 = 3000 ms) exceeds the 2500 ms turn budget, so fix it by streaming LLM tokens directly into TTS (and/or using a faster model) so the stages overlap and end-to-end latency drops below the timeout.",
            "content_chars": 237,
            "content_excerpt": "The serialized pipeline (550 + 1800 + 650 = 3000 ms) exceeds the 2500 ms turn budget, so fix it by streaming LLM tokens directly into TTS (and/or using a faster model) so the stages overlap and end-to",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 4786.082175000047,
            "reasoning_chars": 1233,
            "reasoning_excerpt": "The user gives a scenario: voice agent pipeline with STT, LLM, TTS. Total turn timeout is 2500 ms. Component latencies: STT=550, LLM=1800, TTS=650. Sum = 550+1800+650 = 3000 ms, which exceeds the 2500",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "identifies_budget_overrun",
                "passed": true
              },
              {
                "name": "offers_latency_fix",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 367,
              "prompt_tokens": 71,
              "prompt_tokens_details": null,
              "total_tokens": 438
            }
          },
          {
            "attempt": 3,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 6144,
              "reasoning_headroom_tokens": 4096,
              "visible_answer_tokens": 2048
            },
            "content": "The pipeline's serial latency (550+1800+650 = 3000 ms) exceeds the 2500 ms turn timeout by 500 ms, so turns will fail or get cut off—fix it by streaming end-to-end (partial STT results → token-by-token LLM → chunked TTS) so first audio plays before the full LLM response completes, or swap in a faster/smaller LLM to cut the largest latency contributor.",
            "content_chars": 353,
            "content_excerpt": "The pipeline's serial latency (550+1800+650 = 3000 ms) exceeds the 2500 ms turn timeout by 500 ms, so turns will fail or get cut off—fix it by streaming end-to-end (partial STT results → token-by-toke",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 4382.0809589960845,
            "reasoning_chars": 970,
            "reasoning_excerpt": "The user asks about a voice agent pipeline: STT (speech-to-text), LLM, TTS (text-to-speech). Total turn timeout is 2500 ms. Component latencies: STT=550, LLM=1800, TTS=650. Sum = 550 + 1800 + 650 = 30",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "identifies_budget_overrun",
                "passed": true
              },
              {
                "name": "offers_latency_fix",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 335,
              "prompt_tokens": 71,
              "prompt_tokens_details": null,
              "total_tokens": 406
            }
          }
        ],
        "id": "parallel_timeout_triage",
        "pass_count": 3,
        "pass_rate": 1.0,
        "required_pass_rate": 1.0,
        "status": "passed",
        "validator": "deterministic_text_checks"
      }
    ],
    "status": "passed"
  },
  "run_id": "fast-bakeoff-20260923T223524Z",
  "sampling": {
    "temperature": {
      "effective_request": 0.6,
      "requested": 0.6,
      "sent": true
    },
    "top_p": {
      "effective_request": 0.95,
      "requested": 0.95,
      "sent": true
    }
  },
  "schema": "anvil-serving.fast-tier-bakeoff/v1",
  "score_inputs": {
    "e2e_p50_ms": null,
    "intelligence_pass_rate": 1.0,
    "operational_fit_notes": [
      "endpoint was already loaded; benchmark did not start or stop serves"
    ],
    "session_recall_passed": true,
    "thinking_mode": "default",
    "tool_call_passed": true,
    "ttft_p50_ms": null,
    "usable_context_tokens": null,
    "voice_latency_ms": null
  },
  "selection": {
    "context_targets": [
      32768
    ],
    "endpoint_already_loaded": true,
    "requests_per_context": 1,
    "suites": [
      "intelligence",
      "tool",
      "session"
    ]
  },
  "session": {
    "checks": [
      {
        "attempt_count": 3,
        "attempts": [
          {
            "attempt": 1,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 6144,
              "reasoning_headroom_tokens": 4096,
              "visible_answer_tokens": 2048
            },
            "content": "RIVER-918",
            "content_chars": 9,
            "content_excerpt": "RIVER-918",
            "expected": "RIVER-918",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 670.7448070010287,
            "reasoning_chars": 132,
            "reasoning_excerpt": "The user asked me to remember a session code \"RIVER-918\" and now asks what session code should be used, replying with only the code.",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "usage": {
              "completion_tokens": 40,
              "prompt_tokens": 45,
              "prompt_tokens_details": null,
              "total_tokens": 85
            }
          },
          {
            "attempt": 2,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 6144,
              "reasoning_headroom_tokens": 4096,
              "visible_answer_tokens": 2048
            },
            "content": "RIVER-918",
            "content_chars": 9,
            "content_excerpt": "RIVER-918",
            "expected": "RIVER-918",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 914.4131249995553,
            "reasoning_chars": 226,
            "reasoning_excerpt": "The user asked me to remember the session code RIVER-918, and I replied \"ok\". Now they're asking what session code should be used, and to reply with only the code.\n\nThis is a simple memory test - I sh",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "usage": {
              "completion_tokens": 64,
              "prompt_tokens": 45,
              "prompt_tokens_details": null,
              "total_tokens": 109
            }
          },
          {
            "attempt": 3,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 6144,
              "reasoning_headroom_tokens": 4096,
              "visible_answer_tokens": 2048
            },
            "content": "RIVER-918",
            "content_chars": 9,
            "content_excerpt": "RIVER-918",
            "expected": "RIVER-918",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 914.8539259986137,
            "reasoning_chars": 228,
            "reasoning_excerpt": "The user asked me to remember the session code RIVER-918, and I replied \"ok\". Now they're asking what session code should be used, and to reply with only the code.\n\nThis is a simple memory recall requ",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "usage": {
              "completion_tokens": 64,
              "prompt_tokens": 45,
              "prompt_tokens_details": null,
              "total_tokens": 109
            }
          }
        ],
        "name": "single_request_multiturn_recall",
        "pass_count": 3,
        "pass_rate": 1.0,
        "required_pass_rate": 1.0,
        "status": "passed"
      }
    ],
    "status": "passed"
  },
  "source_recipe": {
    "ref": null,
    "serve_command": null
  },
  "suites": {},
  "thinking": {
    "chat_template_kwargs": null,
    "control_evidence": null,
    "control_evidence_sha256": null,
    "control_mechanism": "reasoning_effort",
    "control_requested": "max",
    "control_status": "requested_unverified",
    "mode": "default",
    "reasoning_effort": "max",
    "unsupported": false
  },
  "timing": {
    "chat": {
      "completion_tokens_max": null,
      "completion_tokens_mean": null,
      "completion_tokens_mean_ci95_high": null,
      "completion_tokens_mean_ci95_low": null,
      "completion_tokens_min": null,
      "completion_tokens_p25": null,
      "completion_tokens_p50": null,
      "completion_tokens_p75": null,
      "completion_tokens_p90": null,
      "completion_tokens_p95": null,
      "completion_tokens_p99": null,
      "completion_tokens_samples": 0,
      "completion_tokens_stddev": null,
      "content_chunks": 0,
      "controlled_output_exact_adherence": 0,
      "controlled_output_exact_adherence_rate": null,
      "controlled_output_samples": 0,
      "controlled_output_unobservable": 0,
      "decode_tok_s_max": null,
      "decode_tok_s_mean": null,
      "decode_tok_s_mean_ci95_high": null,
      "decode_tok_s_mean_ci95_low": null,
      "decode_tok_s_min": null,
      "decode_tok_s_p25": null,
      "decode_tok_s_p50": null,
      "decode_tok_s_p75": null,
      "decode_tok_s_p90": null,
      "decode_tok_s_p95": null,
      "decode_tok_s_p99": null,
      "decode_tok_s_samples": 0,
      "decode_tok_s_stddev": null,
      "e2e_max_ms": null,
      "e2e_mean_ci95_high_ms": null,
      "e2e_mean_ci95_low_ms": null,
      "e2e_mean_ms": null,
      "e2e_min_ms": null,
      "e2e_p25_ms": null,
      "e2e_p50_ms": null,
      "e2e_p75_ms": null,
      "e2e_p90_ms": null,
      "e2e_p95_ms": null,
      "e2e_p99_ms": null,
      "e2e_samples": 0,
      "e2e_stddev_ms": null,
      "effective_prefill_tok_s_max": null,
      "effective_prefill_tok_s_mean": null,
      "effective_prefill_tok_s_mean_ci95_high": null,
      "effective_prefill_tok_s_mean_ci95_low": null,
      "effective_prefill_tok_s_min": null,
      "effective_prefill_tok_s_p25": null,
      "effective_prefill_tok_s_p50": null,
      "effective_prefill_tok_s_p75": null,
      "effective_prefill_tok_s_p90": null,
      "effective_prefill_tok_s_p95": null,
      "effective_prefill_tok_s_p99": null,
      "effective_prefill_tok_s_samples": 0,
      "effective_prefill_tok_s_stddev": null,
      "generation_max_ms": null,
      "generation_mean_ci95_high_ms": null,
      "generation_mean_ci95_low_ms": null,
      "generation_mean_ms": null,
      "generation_min_ms": null,
      "generation_p25_ms": null,
      "generation_p50_ms": null,
      "generation_p75_ms": null,
      "generation_p90_ms": null,
      "generation_p95_ms": null,
      "generation_p99_ms": null,
      "generation_samples": 0,
      "generation_stddev_ms": null,
      "mean_inter_token_latency_ms_max": null,
      "mean_inter_token_latency_ms_mean": null,
      "mean_inter_token_latency_ms_mean_ci95_high": null,
      "mean_inter_token_latency_ms_mean_ci95_low": null,
      "mean_inter_token_latency_ms_min": null,
      "mean_inter_token_latency_ms_p25": null,
      "mean_inter_token_latency_ms_p50": null,
      "mean_inter_token_latency_ms_p75": null,
      "mean_inter_token_latency_ms_p90": null,
      "mean_inter_token_latency_ms_p95": null,
      "mean_inter_token_latency_ms_p99": null,
      "mean_inter_token_latency_ms_samples": 0,
      "mean_inter_token_latency_ms_stddev": null,
      "mean_time_per_output_token_max_ms": null,
      "mean_time_per_output_token_mean_ci95_high_ms": null,
      "mean_time_per_output_token_mean_ci95_low_ms": null,
      "mean_time_per_output_token_mean_ms": null,
      "mean_time_per_output_token_min_ms": null,
      "mean_time_per_output_token_p25_ms": null,
      "mean_time_per_output_token_p50_ms": null,
      "mean_time_per_output_token_p75_ms": null,
      "mean_time_per_output_token_p90_ms": null,
      "mean_time_per_output_token_p95_ms": null,
      "mean_time_per_output_token_p99_ms": null,
      "mean_time_per_output_token_samples": 0,
      "mean_time_per_output_token_stddev_ms": null,
      "output_tokens": 0,
      "prompt_token_samples": 0,
      "prompt_tokens": null,
      "prompt_tokens_max": null,
      "prompt_tokens_mean": null,
      "prompt_tokens_mean_ci95_high": null,
      "prompt_tokens_mean_ci95_low": null,
      "prompt_tokens_min": null,
      "prompt_tokens_p25": null,
      "prompt_tokens_p50": null,
      "prompt_tokens_p75": null,
      "prompt_tokens_p90": null,
      "prompt_tokens_p95": null,
      "prompt_tokens_p99": null,
      "prompt_tokens_samples": 0,
      "prompt_tokens_stddev": null,
      "reasoning_chunks": 0,
      "request_canary_passed": 0,
      "request_canary_samples": 0,
      "time_to_first_output_max_ms": null,
      "time_to_first_output_mean_ci95_high_ms": null,
      "time_to_first_output_mean_ci95_low_ms": null,
      "time_to_first_output_mean_ms": null,
      "time_to_first_output_min_ms": null,
      "time_to_first_output_p25_ms": null,
      "time_to_first_output_p50_ms": null,
      "time_to_first_output_p75_ms": null,
      "time_to_first_output_p90_ms": null,
      "time_to_first_output_p95_ms": null,
      "time_to_first_output_p99_ms": null,
      "time_to_first_output_samples": 0,
      "time_to_first_output_stddev_ms": null,
      "tpot_max_ms": null,
      "tpot_mean_ci95_high_ms": null,
      "tpot_mean_ci95_low_ms": null,
      "tpot_mean_ms": null,
      "tpot_min_ms": null,
      "tpot_p25_ms": null,
      "tpot_p50_ms": null,
      "tpot_p75_ms": null,
      "tpot_p90_ms": null,
      "tpot_p95_ms": null,
      "tpot_p99_ms": null,
      "tpot_samples": 0,
      "tpot_stddev_ms": null,
      "ttft_max_ms": null,
      "ttft_mean_ci95_high_ms": null,
      "ttft_mean_ci95_low_ms": null,
      "ttft_mean_ms": null,
      "ttft_min_ms": null,
      "ttft_p25_ms": null,
      "ttft_p50_ms": null,
      "ttft_p75_ms": null,
      "ttft_p90_ms": null,
      "ttft_p95_ms": null,
      "ttft_p99_ms": null,
      "ttft_samples": 0,
      "ttft_stddev_ms": null
    },
    "wall_ms": 45954.881618999934
  },
  "tool": {
    "checks": [
      {
        "attempt_count": 3,
        "attempts": [
          {
            "arguments": {
              "zip": "98101"
            },
            "attempt": 1,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 6144,
              "reasoning_headroom_tokens": 4096,
              "visible_answer_tokens": 2048
            },
            "content": "",
            "content_chars": 0,
            "content_excerpt": "",
            "failure_class": null,
            "finish_reason": "tool_calls",
            "latency_ms": 577.65417900373,
            "reasoning_chars": 78,
            "reasoning_excerpt": "The user wants me to call record_weather_zip with zip 98101. Simple tool call.",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "tool_argument_observations": [
              {
                "arguments_chars": 16,
                "arguments_excerpt": "{\"zip\": \"98101\"}",
                "arguments_truncated": false,
                "arguments_wire_type": "str",
                "name": "record_weather_zip"
              }
            ],
            "tool_call_count": 1,
            "usage": {
              "completion_tokens": 35,
              "prompt_tokens": 171,
              "prompt_tokens_details": null,
              "total_tokens": 206
            },
            "valid_tool_call_count": 1,
            "validation_errors": []
          },
          {
            "arguments": {
              "zip": "98101"
            },
            "attempt": 2,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 6144,
              "reasoning_headroom_tokens": 4096,
              "visible_answer_tokens": 2048
            },
            "content": "",
            "content_chars": 0,
            "content_excerpt": "",
            "failure_class": null,
            "finish_reason": "tool_calls",
            "latency_ms": 571.944422001252,
            "reasoning_chars": 85,
            "reasoning_excerpt": "The user wants me to call record_weather_zip with zip 98101. Simple single tool call.",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "tool_argument_observations": [
              {
                "arguments_chars": 16,
                "arguments_excerpt": "{\"zip\": \"98101\"}",
                "arguments_truncated": false,
                "arguments_wire_type": "str",
                "name": "record_weather_zip"
              }
            ],
            "tool_call_count": 1,
            "usage": {
              "completion_tokens": 36,
              "prompt_tokens": 171,
              "prompt_tokens_details": null,
              "total_tokens": 207
            },
            "valid_tool_call_count": 1,
            "validation_errors": []
          },
          {
            "arguments": {
              "zip": "98101"
            },
            "attempt": 3,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 6144,
              "reasoning_headroom_tokens": 4096,
              "visible_answer_tokens": 2048
            },
            "content": "",
            "content_chars": 0,
            "content_excerpt": "",
            "failure_class": null,
            "finish_reason": "tool_calls",
            "latency_ms": 562.2333459978108,
            "reasoning_chars": 78,
            "reasoning_excerpt": "The user wants me to call record_weather_zip with zip 98101. Simple tool call.",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "tool_argument_observations": [
              {
                "arguments_chars": 16,
                "arguments_excerpt": "{\"zip\": \"98101\"}",
                "arguments_truncated": false,
                "arguments_wire_type": "str",
                "name": "record_weather_zip"
              }
            ],
            "tool_call_count": 1,
            "usage": {
              "completion_tokens": 35,
              "prompt_tokens": 171,
              "prompt_tokens_details": null,
              "total_tokens": 206
            },
            "valid_tool_call_count": 1,
            "validation_errors": []
          }
        ],
        "expected_arguments": {
          "zip": "98101"
        },
        "expected_function": "record_weather_zip",
        "name": "openai_tool_call_smoke",
        "pass_count": 3,
        "pass_rate": 1.0,
        "required_pass_rate": 1.0,
        "status": "passed"
      }
    ],
    "status": "passed"
  },
  "voice": {
    "status": "not_run",
    "stt_latency_ms": null,
    "total_turn_latency_ms": null,
    "tts_latency_ms": null
  }
}
