{
  "context": {
    "cap_tokens": 317440,
    "max_model_len": 327680,
    "targets": []
  },
  "evaluation_protocol": {
    "budget_semantics": "visible_answer_tokens plus reasoning_headroom_tokens are sent as one max_tokens cap; the endpoint does not hard-partition the channels",
    "minimum_comparable_repetitions": 3,
    "minimum_pass_rate": 1.0,
    "reasoning_headroom_tokens": 8192,
    "records_finish_reason": true,
    "records_full_visible_answer": true,
    "records_reasoning_channel_metadata": true,
    "repetitions": 3,
    "version": 3,
    "visible_answer_tokens": 1024
  },
  "failures": [],
  "identity": {
    "base_url": "http://127.0.0.1:30000/v1",
    "candidate_id": "glm53-control",
    "config_id": "r9",
    "engine": "vllm-v84",
    "gpu": "2x-rtx-pro-6000-max-q",
    "model": "glm53-flash-exl3-4bpw-v84-fp8-tp2-c4-327k-r9-vision8-nospec",
    "started_at": "2026-09-19T11:55:41Z"
  },
  "intelligence": {
    "checks": [
      {
        "attempt_count": 3,
        "attempts": [
          {
            "attempt": 1,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 9216,
              "reasoning_headroom_tokens": 8192,
              "visible_answer_tokens": 1024
            },
            "content": "```diff\n--- a/app.py\n+++ b/app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "content_chars": 94,
            "content_excerpt": "```diff\n--- a/app.py\n+++ b/app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 4886.554123018868,
            "reasoning_chars": 1093,
            "reasoning_excerpt": "The user wants a unified diff that changes `timeout = 30` to `timeout = 45` while leaving `retries = 2` unchanged.\n\nThe original file:\n```\ntimeout = 30\nretries = 2\n```\n\nA unified diff format looks lik",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "diff_shape",
                "passed": true
              },
              {
                "name": "removes_old_timeout",
                "passed": true
              },
              {
                "name": "adds_new_timeout",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 357,
              "prompt_tokens": 48,
              "prompt_tokens_details": null,
              "total_tokens": 405
            }
          },
          {
            "attempt": 2,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 9216,
              "reasoning_headroom_tokens": 8192,
              "visible_answer_tokens": 1024
            },
            "content": "```diff\n--- app.py\n+++ app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "content_chars": 90,
            "content_excerpt": "```diff\n--- app.py\n+++ app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 6211.991858086549,
            "reasoning_chars": 1378,
            "reasoning_excerpt": "The user wants me to edit app.py, changing timeout from 30 to 45, leaving retries unchanged. They want only a unified diff.\n\nOriginal file:\n```\ntimeout = 30\nretries = 2\n```\n\nA unified diff format:\n```",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "diff_shape",
                "passed": true
              },
              {
                "name": "removes_old_timeout",
                "passed": true
              },
              {
                "name": "adds_new_timeout",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 456,
              "prompt_tokens": 48,
              "prompt_tokens_details": null,
              "total_tokens": 504
            }
          },
          {
            "attempt": 3,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 9216,
              "reasoning_headroom_tokens": 8192,
              "visible_answer_tokens": 1024
            },
            "content": "```diff\n--- a/app.py\n+++ b/app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "content_chars": 94,
            "content_excerpt": "```diff\n--- a/app.py\n+++ b/app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 6145.924739888869,
            "reasoning_chars": 1358,
            "reasoning_excerpt": "The user wants a unified diff that changes `timeout = 30` to `timeout = 45` while leaving `retries = 2` unchanged.\n\nA unified diff format looks like:\n\n```\n--- a/app.py\n+++ b/app.py\n@@ -1,2 +1,2 @@\n ti",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "diff_shape",
                "passed": true
              },
              {
                "name": "removes_old_timeout",
                "passed": true
              },
              {
                "name": "adds_new_timeout",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 451,
              "prompt_tokens": 48,
              "prompt_tokens_details": null,
              "total_tokens": 499
            }
          }
        ],
        "id": "unified_diff_timeout_edit",
        "pass_count": 3,
        "pass_rate": 1.0,
        "required_pass_rate": 1.0,
        "status": "passed",
        "validator": "deterministic_text_checks"
      },
      {
        "attempt_count": 3,
        "attempts": [
          {
            "attempt": 1,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 9216,
              "reasoning_headroom_tokens": 8192,
              "visible_answer_tokens": 1024
            },
            "content": "The serial pipeline latency (550 + 1800 + 650 = 3000 ms) exceeds the 2500 ms turn budget by 500 ms, with the LLM dominating; a practical fix is to stream LLM tokens directly into TTS (and STT partials into the LLM) so stages overlap instead of running back-to-back.",
            "content_chars": 265,
            "content_excerpt": "The serial pipeline latency (550 + 1800 + 650 = 3000 ms) exceeds the 2500 ms turn budget by 500 ms, with the LLM dominating; a practical fix is to stream LLM tokens directly into TTS (and STT partials",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 5673.504525911994,
            "reasoning_chars": 1397,
            "reasoning_excerpt": "The user asks about a voice agent pipeline: STT (speech-to-text), LLM, TTS (text-to-speech). Total turn timeout is 2500 ms. Logs: STT=550, LLM=1800, TTS=650. Sum = 3000 ms, which exceeds the 2500 ms t",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "identifies_budget_overrun",
                "passed": true
              },
              {
                "name": "offers_latency_fix",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 416,
              "prompt_tokens": 71,
              "prompt_tokens_details": null,
              "total_tokens": 487
            }
          },
          {
            "attempt": 2,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 9216,
              "reasoning_headroom_tokens": 8192,
              "visible_answer_tokens": 1024
            },
            "content": "The serial pipeline latency (550 + 1800 + 650 = 3000 ms) exceeds the 2500 ms turn budget, with the LLM consuming 72% of it, so the practical fix is to stream LLM tokens incrementally into TTS (and/or switch to a smaller/faster model) so audio playback begins before generation finishes.",
            "content_chars": 286,
            "content_excerpt": "The serial pipeline latency (550 + 1800 + 650 = 3000 ms) exceeds the 2500 ms turn budget, with the LLM consuming 72% of it, so the practical fix is to stream LLM tokens incrementally into TTS (and/or ",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 5993.966010981239,
            "reasoning_chars": 1494,
            "reasoning_excerpt": "The user is asking about a voice agent pipeline: STT (speech-to-text), LLM, TTS (text-to-speech). Total turn timeout is 2500 ms. The individual latencies: STT=550, LLM=1800, TTS=650. Sum = 550 + 1800 ",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "identifies_budget_overrun",
                "passed": true
              },
              {
                "name": "offers_latency_fix",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 440,
              "prompt_tokens": 71,
              "prompt_tokens_details": null,
              "total_tokens": 511
            }
          },
          {
            "attempt": 3,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 9216,
              "reasoning_headroom_tokens": 8192,
              "visible_answer_tokens": 1024
            },
            "content": "The pipeline's serialized latency (550+1800+650 = 3000 ms) blows the 2500 ms budget by 500 ms, with the LLM dominating at 1800 ms \u2014 so stream LLM tokens directly into TTS (with streaming STT) so first audio plays before generation finishes, or switch to a faster/smaller LLM to cut that 1800 ms down.",
            "content_chars": 300,
            "content_excerpt": "The pipeline's serialized latency (550+1800+650 = 3000 ms) blows the 2500 ms budget by 500 ms, with the LLM dominating at 1800 ms \u2014 so stream LLM tokens directly into TTS (with streaming STT) so first",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 5549.212095094845,
            "reasoning_chars": 1248,
            "reasoning_excerpt": "The user asks about a voice agent pipeline: STT (speech-to-text), LLM, TTS (text-to-speech). Total turn timeout is 2500 ms. Logs: STT=550, LLM=1800, TTS=650. Sum = 550+1800+650 = 3000 ms, which exceed",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "identifies_budget_overrun",
                "passed": true
              },
              {
                "name": "offers_latency_fix",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 407,
              "prompt_tokens": 71,
              "prompt_tokens_details": null,
              "total_tokens": 478
            }
          }
        ],
        "id": "parallel_timeout_triage",
        "pass_count": 3,
        "pass_rate": 1.0,
        "required_pass_rate": 1.0,
        "status": "passed",
        "validator": "deterministic_text_checks"
      }
    ],
    "status": "passed"
  },
  "run_id": "fast-bakeoff-20260919T115541Z",
  "sampling": {
    "temperature": {
      "effective_request": 0.6,
      "requested": 0.6,
      "sent": true
    },
    "top_p": {
      "effective_request": 0.95,
      "requested": 0.95,
      "sent": true
    }
  },
  "schema": "anvil-serving.fast-tier-bakeoff/v1",
  "score_inputs": {
    "e2e_p50_ms": null,
    "intelligence_pass_rate": 1.0,
    "operational_fit_notes": [
      "endpoint was already loaded; benchmark did not start or stop serves"
    ],
    "session_recall_passed": true,
    "thinking_mode": "default",
    "tool_call_passed": true,
    "ttft_p50_ms": null,
    "usable_context_tokens": null,
    "voice_latency_ms": null
  },
  "selection": {
    "context_targets": [
      32768
    ],
    "endpoint_already_loaded": true,
    "requests_per_context": 1,
    "suites": [
      "intelligence",
      "tool",
      "session"
    ]
  },
  "session": {
    "checks": [
      {
        "attempt_count": 3,
        "attempts": [
          {
            "attempt": 1,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 9216,
              "reasoning_headroom_tokens": 8192,
              "visible_answer_tokens": 1024
            },
            "content": "RIVER-918",
            "content_chars": 9,
            "content_excerpt": "RIVER-918",
            "expected": "RIVER-918",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 875.3961430629715,
            "reasoning_chars": 196,
            "reasoning_excerpt": "The user asked me to remember a session code \"RIVER-918\" and now asks what session code should be used, replying with only the code.\n\nThis is a simple memory test. I should reply with \"RIVER-918\".",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "usage": {
              "completion_tokens": 58,
              "prompt_tokens": 45,
              "prompt_tokens_details": null,
              "total_tokens": 103
            }
          },
          {
            "attempt": 2,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 9216,
              "reasoning_headroom_tokens": 8192,
              "visible_answer_tokens": 1024
            },
            "content": "RIVER-918",
            "content_chars": 9,
            "content_excerpt": "RIVER-918",
            "expected": "RIVER-918",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 1250.0593409640715,
            "reasoning_chars": 310,
            "reasoning_excerpt": "The user asked me to remember a session code: RIVER-918. Then they asked what session code should be used, and to reply with only the code.\n\nThis is a simple memory test. The user told me to remember ",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "usage": {
              "completion_tokens": 86,
              "prompt_tokens": 45,
              "prompt_tokens_details": null,
              "total_tokens": 131
            }
          },
          {
            "attempt": 3,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 9216,
              "reasoning_headroom_tokens": 8192,
              "visible_answer_tokens": 1024
            },
            "content": "RIVER-918",
            "content_chars": 9,
            "content_excerpt": "RIVER-918",
            "expected": "RIVER-918",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 1090.0751689914614,
            "reasoning_chars": 273,
            "reasoning_excerpt": "The user asked me to remember the session code RIVER-918, and I replied \"ok\". Now they're asking what session code should be used, and to reply with only the code.\n\nThis is a simple memory recall task",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "usage": {
              "completion_tokens": 74,
              "prompt_tokens": 45,
              "prompt_tokens_details": null,
              "total_tokens": 119
            }
          }
        ],
        "name": "single_request_multiturn_recall",
        "pass_count": 3,
        "pass_rate": 1.0,
        "required_pass_rate": 1.0,
        "status": "passed"
      }
    ],
    "status": "passed"
  },
  "source_recipe": {
    "ref": null,
    "serve_command": null
  },
  "suites": {},
  "thinking": {
    "chat_template_kwargs": null,
    "control_evidence": "reasoning-control.json",
    "control_evidence_sha256": "4071a778733fa9dba47c6f8bcd2a2b81ae5fbe0763e60601a687436ee64ac553",
    "control_mechanism": "reasoning_effort",
    "control_requested": "max",
    "control_status": "verified",
    "mode": "default",
    "reasoning_effort": "max",
    "unsupported": false
  },
  "timing": {
    "chat": {
      "completion_tokens_max": null,
      "completion_tokens_mean": null,
      "completion_tokens_mean_ci95_high": null,
      "completion_tokens_mean_ci95_low": null,
      "completion_tokens_min": null,
      "completion_tokens_p25": null,
      "completion_tokens_p50": null,
      "completion_tokens_p75": null,
      "completion_tokens_p90": null,
      "completion_tokens_p95": null,
      "completion_tokens_p99": null,
      "completion_tokens_samples": 0,
      "completion_tokens_stddev": null,
      "content_chunks": 0,
      "controlled_output_exact_adherence": 0,
      "controlled_output_exact_adherence_rate": null,
      "controlled_output_samples": 0,
      "controlled_output_unobservable": 0,
      "decode_tok_s_max": null,
      "decode_tok_s_mean": null,
      "decode_tok_s_mean_ci95_high": null,
      "decode_tok_s_mean_ci95_low": null,
      "decode_tok_s_min": null,
      "decode_tok_s_p25": null,
      "decode_tok_s_p50": null,
      "decode_tok_s_p75": null,
      "decode_tok_s_p90": null,
      "decode_tok_s_p95": null,
      "decode_tok_s_p99": null,
      "decode_tok_s_samples": 0,
      "decode_tok_s_stddev": null,
      "e2e_max_ms": null,
      "e2e_mean_ci95_high_ms": null,
      "e2e_mean_ci95_low_ms": null,
      "e2e_mean_ms": null,
      "e2e_min_ms": null,
      "e2e_p25_ms": null,
      "e2e_p50_ms": null,
      "e2e_p75_ms": null,
      "e2e_p90_ms": null,
      "e2e_p95_ms": null,
      "e2e_p99_ms": null,
      "e2e_samples": 0,
      "e2e_stddev_ms": null,
      "effective_prefill_tok_s_max": null,
      "effective_prefill_tok_s_mean": null,
      "effective_prefill_tok_s_mean_ci95_high": null,
      "effective_prefill_tok_s_mean_ci95_low": null,
      "effective_prefill_tok_s_min": null,
      "effective_prefill_tok_s_p25": null,
      "effective_prefill_tok_s_p50": null,
      "effective_prefill_tok_s_p75": null,
      "effective_prefill_tok_s_p90": null,
      "effective_prefill_tok_s_p95": null,
      "effective_prefill_tok_s_p99": null,
      "effective_prefill_tok_s_samples": 0,
      "effective_prefill_tok_s_stddev": null,
      "generation_max_ms": null,
      "generation_mean_ci95_high_ms": null,
      "generation_mean_ci95_low_ms": null,
      "generation_mean_ms": null,
      "generation_min_ms": null,
      "generation_p25_ms": null,
      "generation_p50_ms": null,
      "generation_p75_ms": null,
      "generation_p90_ms": null,
      "generation_p95_ms": null,
      "generation_p99_ms": null,
      "generation_samples": 0,
      "generation_stddev_ms": null,
      "mean_inter_token_latency_ms_max": null,
      "mean_inter_token_latency_ms_mean": null,
      "mean_inter_token_latency_ms_mean_ci95_high": null,
      "mean_inter_token_latency_ms_mean_ci95_low": null,
      "mean_inter_token_latency_ms_min": null,
      "mean_inter_token_latency_ms_p25": null,
      "mean_inter_token_latency_ms_p50": null,
      "mean_inter_token_latency_ms_p75": null,
      "mean_inter_token_latency_ms_p90": null,
      "mean_inter_token_latency_ms_p95": null,
      "mean_inter_token_latency_ms_p99": null,
      "mean_inter_token_latency_ms_samples": 0,
      "mean_inter_token_latency_ms_stddev": null,
      "mean_time_per_output_token_max_ms": null,
      "mean_time_per_output_token_mean_ci95_high_ms": null,
      "mean_time_per_output_token_mean_ci95_low_ms": null,
      "mean_time_per_output_token_mean_ms": null,
      "mean_time_per_output_token_min_ms": null,
      "mean_time_per_output_token_p25_ms": null,
      "mean_time_per_output_token_p50_ms": null,
      "mean_time_per_output_token_p75_ms": null,
      "mean_time_per_output_token_p90_ms": null,
      "mean_time_per_output_token_p95_ms": null,
      "mean_time_per_output_token_p99_ms": null,
      "mean_time_per_output_token_samples": 0,
      "mean_time_per_output_token_stddev_ms": null,
      "output_tokens": 0,
      "prompt_token_samples": 0,
      "prompt_tokens": null,
      "prompt_tokens_max": null,
      "prompt_tokens_mean": null,
      "prompt_tokens_mean_ci95_high": null,
      "prompt_tokens_mean_ci95_low": null,
      "prompt_tokens_min": null,
      "prompt_tokens_p25": null,
      "prompt_tokens_p50": null,
      "prompt_tokens_p75": null,
      "prompt_tokens_p90": null,
      "prompt_tokens_p95": null,
      "prompt_tokens_p99": null,
      "prompt_tokens_samples": 0,
      "prompt_tokens_stddev": null,
      "reasoning_chunks": 0,
      "request_canary_passed": 0,
      "request_canary_samples": 0,
      "time_to_first_output_max_ms": null,
      "time_to_first_output_mean_ci95_high_ms": null,
      "time_to_first_output_mean_ci95_low_ms": null,
      "time_to_first_output_mean_ms": null,
      "time_to_first_output_min_ms": null,
      "time_to_first_output_p25_ms": null,
      "time_to_first_output_p50_ms": null,
      "time_to_first_output_p75_ms": null,
      "time_to_first_output_p90_ms": null,
      "time_to_first_output_p95_ms": null,
      "time_to_first_output_p99_ms": null,
      "time_to_first_output_samples": 0,
      "time_to_first_output_stddev_ms": null,
      "tpot_max_ms": null,
      "tpot_mean_ci95_high_ms": null,
      "tpot_mean_ci95_low_ms": null,
      "tpot_mean_ms": null,
      "tpot_min_ms": null,
      "tpot_p25_ms": null,
      "tpot_p50_ms": null,
      "tpot_p75_ms": null,
      "tpot_p90_ms": null,
      "tpot_p95_ms": null,
      "tpot_p99_ms": null,
      "tpot_samples": 0,
      "tpot_stddev_ms": null,
      "ttft_max_ms": null,
      "ttft_mean_ci95_high_ms": null,
      "ttft_mean_ci95_low_ms": null,
      "ttft_mean_ms": null,
      "ttft_min_ms": null,
      "ttft_p25_ms": null,
      "ttft_p50_ms": null,
      "ttft_p75_ms": null,
      "ttft_p90_ms": null,
      "ttft_p95_ms": null,
      "ttft_p99_ms": null,
      "ttft_samples": 0,
      "ttft_stddev_ms": null
    },
    "wall_ms": 39475.69064900745
  },
  "tool": {
    "checks": [
      {
        "attempt_count": 3,
        "attempts": [
          {
            "arguments": {
              "zip": "98101"
            },
            "attempt": 1,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 9216,
              "reasoning_headroom_tokens": 8192,
              "visible_answer_tokens": 1024
            },
            "content": "",
            "content_chars": 0,
            "content_excerpt": "",
            "failure_class": null,
            "finish_reason": "tool_calls",
            "latency_ms": 593.2002359768376,
            "reasoning_chars": 78,
            "reasoning_excerpt": "The user wants me to call record_weather_zip with zip 98101. Simple tool call.",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "tool_call_count": 1,
            "usage": {
              "completion_tokens": 35,
              "prompt_tokens": 171,
              "prompt_tokens_details": null,
              "total_tokens": 206
            },
            "valid_tool_call_count": 1,
            "validation_errors": []
          },
          {
            "arguments": {
              "zip": "98101"
            },
            "attempt": 2,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 9216,
              "reasoning_headroom_tokens": 8192,
              "visible_answer_tokens": 1024
            },
            "content": "",
            "content_chars": 0,
            "content_excerpt": "",
            "failure_class": null,
            "finish_reason": "tool_calls",
            "latency_ms": 600.9244880406186,
            "reasoning_chars": 85,
            "reasoning_excerpt": "The user wants me to call record_weather_zip with zip 98101. Simple single tool call.",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "tool_call_count": 1,
            "usage": {
              "completion_tokens": 36,
              "prompt_tokens": 171,
              "prompt_tokens_details": null,
              "total_tokens": 207
            },
            "valid_tool_call_count": 1,
            "validation_errors": []
          },
          {
            "arguments": {
              "zip": "98101"
            },
            "attempt": 3,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 9216,
              "reasoning_headroom_tokens": 8192,
              "visible_answer_tokens": 1024
            },
            "content": "",
            "content_chars": 0,
            "content_excerpt": "",
            "failure_class": null,
            "finish_reason": "tool_calls",
            "latency_ms": 600.9298469871283,
            "reasoning_chars": 85,
            "reasoning_excerpt": "The user wants me to call record_weather_zip with zip 98101. Simple single tool call.",
            "reasoning_field": "reasoning",
            "reasoning_tokens": null,
            "status": "passed",
            "tool_call_count": 1,
            "usage": {
              "completion_tokens": 36,
              "prompt_tokens": 171,
              "prompt_tokens_details": null,
              "total_tokens": 207
            },
            "valid_tool_call_count": 1,
            "validation_errors": []
          }
        ],
        "expected_arguments": {
          "zip": "98101"
        },
        "expected_function": "record_weather_zip",
        "name": "openai_tool_call_smoke",
        "pass_count": 3,
        "pass_rate": 1.0,
        "required_pass_rate": 1.0,
        "status": "passed"
      }
    ],
    "status": "passed"
  },
  "voice": {
    "status": "not_run",
    "stt_latency_ms": null,
    "total_turn_latency_ms": null,
    "tts_latency_ms": null
  }
}
