{
  "context": {
    "cap_tokens": 2048,
    "max_model_len": 8192,
    "targets": []
  },
  "evaluation_protocol": {
    "budget_semantics": "visible_answer_tokens plus reasoning_headroom_tokens are sent as one max_tokens cap; the endpoint does not hard-partition the channels",
    "minimum_comparable_repetitions": 3,
    "minimum_pass_rate": 1.0,
    "reasoning_headroom_tokens": 4096,
    "records_finish_reason": true,
    "records_full_visible_answer": true,
    "records_reasoning_channel_metadata": true,
    "repetitions": 3,
    "version": 3,
    "visible_answer_tokens": 1024
  },
  "failures": [
    {
      "error": "missing required string argument: zip",
      "failure_classes": [
        "visible_answer_missing"
      ],
      "pass_rate": 0.0,
      "suite": "tool"
    }
  ],
  "identity": {
    "base_url": "http://127.0.0.1:8081/v1",
    "candidate_id": "huihui-nvfp4",
    "config_id": "nospec-8k-reasoning-low-diagnostic",
    "engine": "ninfer",
    "gpu": "rtx5090",
    "model": "qwen38-huihui-ninfer-nvfp4-nospec-8k",
    "started_at": "2026-09-19T18:10:46Z"
  },
  "intelligence": {
    "checks": [
      {
        "attempt_count": 3,
        "attempts": [
          {
            "attempt": 1,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 5120,
              "reasoning_headroom_tokens": 4096,
              "visible_answer_tokens": 1024
            },
            "content": "```\n--- a/app.py\n+++ b/app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "content_chars": 90,
            "content_excerpt": "```\n--- a/app.py\n+++ b/app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 5136.058099975344,
            "reasoning_chars": 958,
            "reasoning_excerpt": "The user wants a unified diff that changes `timeout = 30` to `timeout = 45` while leaving `retries = 2` unchanged.\n\nA unified diff needs:\n- `--- a/app.py` and `+++ b/app.py` headers\n- A hunk header wi",
            "reasoning_field": "reasoning_content",
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "diff_shape",
                "passed": true
              },
              {
                "name": "removes_old_timeout",
                "passed": true
              },
              {
                "name": "adds_new_timeout",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 367,
              "prompt_tokens": 79,
              "total_tokens": 446
            }
          },
          {
            "attempt": 2,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 5120,
              "reasoning_headroom_tokens": 4096,
              "visible_answer_tokens": 1024
            },
            "content": "```\n--- a/app.py\n+++ b/app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "content_chars": 90,
            "content_excerpt": "```\n--- a/app.py\n+++ b/app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 5107.269200030714,
            "reasoning_chars": 958,
            "reasoning_excerpt": "The user wants a unified diff that changes `timeout = 30` to `timeout = 45` while leaving `retries = 2` unchanged.\n\nA unified diff needs:\n- `--- a/app.py` and `+++ b/app.py` headers\n- A hunk header wi",
            "reasoning_field": "reasoning_content",
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "diff_shape",
                "passed": true
              },
              {
                "name": "removes_old_timeout",
                "passed": true
              },
              {
                "name": "adds_new_timeout",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 367,
              "prompt_tokens": 79,
              "total_tokens": 446
            }
          },
          {
            "attempt": 3,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 5120,
              "reasoning_headroom_tokens": 4096,
              "visible_answer_tokens": 1024
            },
            "content": "```\n--- a/app.py\n+++ b/app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "content_chars": 90,
            "content_excerpt": "```\n--- a/app.py\n+++ b/app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 5103.144199994858,
            "reasoning_chars": 958,
            "reasoning_excerpt": "The user wants a unified diff that changes `timeout = 30` to `timeout = 45` while leaving `retries = 2` unchanged.\n\nA unified diff needs:\n- `--- a/app.py` and `+++ b/app.py` headers\n- A hunk header wi",
            "reasoning_field": "reasoning_content",
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "diff_shape",
                "passed": true
              },
              {
                "name": "removes_old_timeout",
                "passed": true
              },
              {
                "name": "adds_new_timeout",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 367,
              "prompt_tokens": 79,
              "total_tokens": 446
            }
          }
        ],
        "id": "unified_diff_timeout_edit",
        "pass_count": 3,
        "pass_rate": 1.0,
        "required_pass_rate": 1.0,
        "status": "passed",
        "validator": "deterministic_text_checks"
      },
      {
        "attempt_count": 3,
        "attempts": [
          {
            "attempt": 1,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 5120,
              "reasoning_headroom_tokens": 4096,
              "visible_answer_tokens": 1024
            },
            "content": "The three stages sum to 3000 ms, exceeding the 2500 ms budget by 500 ms; the practical fix is to stream LLM tokens into TTS so TTS begins on the first sentence rather than waiting for the full 1800 ms LLM response, effectively overlapping the two and bringing total latency under the timeout.",
            "content_chars": 292,
            "content_excerpt": "The three stages sum to 3000 ms, exceeding the 2500 ms budget by 500 ms; the practical fix is to stream LLM tokens into TTS so TTS begins on the first sentence rather than waiting for the full 1800 ms",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 4554.911499959417,
            "reasoning_chars": 798,
            "reasoning_excerpt": "The user wants a single concise sentence that identifies the problem and a practical fix.\n\nLet me analyze:\n- Total turn timeout: 2500 ms\n- STT: 550 ms\n- LLM: 1800 ms\n- TTS: 650 ms\n- Total: 550 + 1800 ",
            "reasoning_field": "reasoning_content",
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "identifies_budget_overrun",
                "passed": true
              },
              {
                "name": "offers_latency_fix",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 326,
              "prompt_tokens": 107,
              "total_tokens": 433
            }
          },
          {
            "attempt": 2,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 5120,
              "reasoning_headroom_tokens": 4096,
              "visible_answer_tokens": 1024
            },
            "content": "The three stages sum to 3000 ms, exceeding the 2500 ms budget by 500 ms; the practical fix is to stream LLM tokens into TTS so TTS begins on the first sentence rather than waiting for the full 1800 ms LLM response, effectively overlapping the two and bringing total latency under the timeout.",
            "content_chars": 292,
            "content_excerpt": "The three stages sum to 3000 ms, exceeding the 2500 ms budget by 500 ms; the practical fix is to stream LLM tokens into TTS so TTS begins on the first sentence rather than waiting for the full 1800 ms",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 4526.178300031461,
            "reasoning_chars": 798,
            "reasoning_excerpt": "The user wants a single concise sentence that identifies the problem and a practical fix.\n\nLet me analyze:\n- Total turn timeout: 2500 ms\n- STT: 550 ms\n- LLM: 1800 ms\n- TTS: 650 ms\n- Total: 550 + 1800 ",
            "reasoning_field": "reasoning_content",
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "identifies_budget_overrun",
                "passed": true
              },
              {
                "name": "offers_latency_fix",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 326,
              "prompt_tokens": 107,
              "total_tokens": 433
            }
          },
          {
            "attempt": 3,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 5120,
              "reasoning_headroom_tokens": 4096,
              "visible_answer_tokens": 1024
            },
            "content": "The three stages sum to 3000 ms, exceeding the 2500 ms budget by 500 ms; the practical fix is to stream LLM tokens into TTS so TTS begins on the first sentence rather than waiting for the full 1800 ms LLM response, effectively overlapping the two and bringing total latency under the timeout.",
            "content_chars": 292,
            "content_excerpt": "The three stages sum to 3000 ms, exceeding the 2500 ms budget by 500 ms; the practical fix is to stream LLM tokens into TTS so TTS begins on the first sentence rather than waiting for the full 1800 ms",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 4526.823799998965,
            "reasoning_chars": 798,
            "reasoning_excerpt": "The user wants a single concise sentence that identifies the problem and a practical fix.\n\nLet me analyze:\n- Total turn timeout: 2500 ms\n- STT: 550 ms\n- LLM: 1800 ms\n- TTS: 650 ms\n- Total: 550 + 1800 ",
            "reasoning_field": "reasoning_content",
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "identifies_budget_overrun",
                "passed": true
              },
              {
                "name": "offers_latency_fix",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 326,
              "prompt_tokens": 107,
              "total_tokens": 433
            }
          }
        ],
        "id": "parallel_timeout_triage",
        "pass_count": 3,
        "pass_rate": 1.0,
        "required_pass_rate": 1.0,
        "status": "passed",
        "validator": "deterministic_text_checks"
      }
    ],
    "status": "passed"
  },
  "run_id": "fast-bakeoff-20260919T181046Z",
  "schema": "anvil-serving.fast-tier-bakeoff/v1",
  "score_inputs": {
    "e2e_p50_ms": null,
    "intelligence_pass_rate": 1.0,
    "operational_fit_notes": [
      "endpoint was already loaded; benchmark did not start or stop serves"
    ],
    "session_recall_passed": false,
    "thinking_mode": "default",
    "tool_call_passed": false,
    "ttft_p50_ms": null,
    "usable_context_tokens": null,
    "voice_latency_ms": null
  },
  "selection": {
    "context_targets": [
      4096
    ],
    "endpoint_already_loaded": true,
    "requests_per_context": 1,
    "suites": [
      "intelligence",
      "tool"
    ]
  },
  "session": {
    "checks": [],
    "status": "not_run"
  },
  "source_recipe": {
    "ref": null,
    "serve_command": null
  },
  "suites": {},
  "thinking": {
    "chat_template_kwargs": null,
    "control_evidence": null,
    "control_evidence_sha256": null,
    "control_mechanism": "reasoning_effort",
    "control_requested": "low",
    "control_status": "requested_unverified",
    "mode": "default",
    "reasoning_effort": "low",
    "unsupported": false
  },
  "timing": {
    "chat": {
      "completion_tokens_max": null,
      "completion_tokens_mean": null,
      "completion_tokens_mean_ci95_high": null,
      "completion_tokens_mean_ci95_low": null,
      "completion_tokens_min": null,
      "completion_tokens_p25": null,
      "completion_tokens_p50": null,
      "completion_tokens_p75": null,
      "completion_tokens_p90": null,
      "completion_tokens_p95": null,
      "completion_tokens_p99": null,
      "completion_tokens_samples": 0,
      "completion_tokens_stddev": null,
      "content_chunks": 0,
      "controlled_output_exact_adherence": 0,
      "controlled_output_exact_adherence_rate": null,
      "controlled_output_samples": 0,
      "controlled_output_unobservable": 0,
      "decode_tok_s_max": null,
      "decode_tok_s_mean": null,
      "decode_tok_s_mean_ci95_high": null,
      "decode_tok_s_mean_ci95_low": null,
      "decode_tok_s_min": null,
      "decode_tok_s_p25": null,
      "decode_tok_s_p50": null,
      "decode_tok_s_p75": null,
      "decode_tok_s_p90": null,
      "decode_tok_s_p95": null,
      "decode_tok_s_p99": null,
      "decode_tok_s_samples": 0,
      "decode_tok_s_stddev": null,
      "e2e_max_ms": null,
      "e2e_mean_ci95_high_ms": null,
      "e2e_mean_ci95_low_ms": null,
      "e2e_mean_ms": null,
      "e2e_min_ms": null,
      "e2e_p25_ms": null,
      "e2e_p50_ms": null,
      "e2e_p75_ms": null,
      "e2e_p90_ms": null,
      "e2e_p95_ms": null,
      "e2e_p99_ms": null,
      "e2e_samples": 0,
      "e2e_stddev_ms": null,
      "effective_prefill_tok_s_max": null,
      "effective_prefill_tok_s_mean": null,
      "effective_prefill_tok_s_mean_ci95_high": null,
      "effective_prefill_tok_s_mean_ci95_low": null,
      "effective_prefill_tok_s_min": null,
      "effective_prefill_tok_s_p25": null,
      "effective_prefill_tok_s_p50": null,
      "effective_prefill_tok_s_p75": null,
      "effective_prefill_tok_s_p90": null,
      "effective_prefill_tok_s_p95": null,
      "effective_prefill_tok_s_p99": null,
      "effective_prefill_tok_s_samples": 0,
      "effective_prefill_tok_s_stddev": null,
      "generation_max_ms": null,
      "generation_mean_ci95_high_ms": null,
      "generation_mean_ci95_low_ms": null,
      "generation_mean_ms": null,
      "generation_min_ms": null,
      "generation_p25_ms": null,
      "generation_p50_ms": null,
      "generation_p75_ms": null,
      "generation_p90_ms": null,
      "generation_p95_ms": null,
      "generation_p99_ms": null,
      "generation_samples": 0,
      "generation_stddev_ms": null,
      "mean_inter_token_latency_ms_max": null,
      "mean_inter_token_latency_ms_mean": null,
      "mean_inter_token_latency_ms_mean_ci95_high": null,
      "mean_inter_token_latency_ms_mean_ci95_low": null,
      "mean_inter_token_latency_ms_min": null,
      "mean_inter_token_latency_ms_p25": null,
      "mean_inter_token_latency_ms_p50": null,
      "mean_inter_token_latency_ms_p75": null,
      "mean_inter_token_latency_ms_p90": null,
      "mean_inter_token_latency_ms_p95": null,
      "mean_inter_token_latency_ms_p99": null,
      "mean_inter_token_latency_ms_samples": 0,
      "mean_inter_token_latency_ms_stddev": null,
      "mean_time_per_output_token_max_ms": null,
      "mean_time_per_output_token_mean_ci95_high_ms": null,
      "mean_time_per_output_token_mean_ci95_low_ms": null,
      "mean_time_per_output_token_mean_ms": null,
      "mean_time_per_output_token_min_ms": null,
      "mean_time_per_output_token_p25_ms": null,
      "mean_time_per_output_token_p50_ms": null,
      "mean_time_per_output_token_p75_ms": null,
      "mean_time_per_output_token_p90_ms": null,
      "mean_time_per_output_token_p95_ms": null,
      "mean_time_per_output_token_p99_ms": null,
      "mean_time_per_output_token_samples": 0,
      "mean_time_per_output_token_stddev_ms": null,
      "output_tokens": 0,
      "prompt_token_samples": 0,
      "prompt_tokens": null,
      "prompt_tokens_max": null,
      "prompt_tokens_mean": null,
      "prompt_tokens_mean_ci95_high": null,
      "prompt_tokens_mean_ci95_low": null,
      "prompt_tokens_min": null,
      "prompt_tokens_p25": null,
      "prompt_tokens_p50": null,
      "prompt_tokens_p75": null,
      "prompt_tokens_p90": null,
      "prompt_tokens_p95": null,
      "prompt_tokens_p99": null,
      "prompt_tokens_samples": 0,
      "prompt_tokens_stddev": null,
      "reasoning_chunks": 0,
      "request_canary_passed": 0,
      "request_canary_samples": 0,
      "time_to_first_output_max_ms": null,
      "time_to_first_output_mean_ci95_high_ms": null,
      "time_to_first_output_mean_ci95_low_ms": null,
      "time_to_first_output_mean_ms": null,
      "time_to_first_output_min_ms": null,
      "time_to_first_output_p25_ms": null,
      "time_to_first_output_p50_ms": null,
      "time_to_first_output_p75_ms": null,
      "time_to_first_output_p90_ms": null,
      "time_to_first_output_p95_ms": null,
      "time_to_first_output_p99_ms": null,
      "time_to_first_output_samples": 0,
      "time_to_first_output_stddev_ms": null,
      "tpot_max_ms": null,
      "tpot_mean_ci95_high_ms": null,
      "tpot_mean_ci95_low_ms": null,
      "tpot_mean_ms": null,
      "tpot_min_ms": null,
      "tpot_p25_ms": null,
      "tpot_p50_ms": null,
      "tpot_p75_ms": null,
      "tpot_p90_ms": null,
      "tpot_p95_ms": null,
      "tpot_p99_ms": null,
      "tpot_samples": 0,
      "tpot_stddev_ms": null,
      "ttft_max_ms": null,
      "ttft_mean_ci95_high_ms": null,
      "ttft_mean_ci95_low_ms": null,
      "ttft_mean_ms": null,
      "ttft_min_ms": null,
      "ttft_p25_ms": null,
      "ttft_p50_ms": null,
      "ttft_p75_ms": null,
      "ttft_p90_ms": null,
      "ttft_p95_ms": null,
      "ttft_p99_ms": null,
      "ttft_samples": 0,
      "ttft_stddev_ms": null
    },
    "wall_ms": 31659.191599988844
  },
  "tool": {
    "checks": [
      {
        "attempt_count": 3,
        "attempts": [
          {
            "arguments": null,
            "attempt": 1,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 5120,
              "reasoning_headroom_tokens": 4096,
              "visible_answer_tokens": 1024
            },
            "content": "",
            "content_chars": 0,
            "content_excerpt": "",
            "error": "missing required string argument: zip",
            "failure_class": "visible_answer_missing",
            "finish_reason": "tool_calls",
            "latency_ms": 1109.429200005252,
            "reasoning_chars": 83,
            "reasoning_excerpt": "The user wants me to call the record_weather_zip function with the zip code 98101.\n",
            "reasoning_field": "reasoning_content",
            "reasoning_tokens": null,
            "status": "failed",
            "tool_call_count": 1,
            "usage": {
              "completion_tokens": 57,
              "prompt_tokens": 316,
              "total_tokens": 373
            },
            "valid_tool_call_count": 0,
            "validation_errors": [
              "missing required string argument: zip"
            ]
          },
          {
            "arguments": null,
            "attempt": 2,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 5120,
              "reasoning_headroom_tokens": 4096,
              "visible_answer_tokens": 1024
            },
            "content": "",
            "content_chars": 0,
            "content_excerpt": "",
            "error": "missing required string argument: zip",
            "failure_class": "visible_answer_missing",
            "finish_reason": "tool_calls",
            "latency_ms": 797.2046000068076,
            "reasoning_chars": 83,
            "reasoning_excerpt": "The user wants me to call the record_weather_zip function with the zip code 98101.\n",
            "reasoning_field": "reasoning_content",
            "reasoning_tokens": null,
            "status": "failed",
            "tool_call_count": 1,
            "usage": {
              "completion_tokens": 57,
              "prompt_tokens": 316,
              "total_tokens": 373
            },
            "valid_tool_call_count": 0,
            "validation_errors": [
              "missing required string argument: zip"
            ]
          },
          {
            "arguments": null,
            "attempt": 3,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 5120,
              "reasoning_headroom_tokens": 4096,
              "visible_answer_tokens": 1024
            },
            "content": "",
            "content_chars": 0,
            "content_excerpt": "",
            "error": "missing required string argument: zip",
            "failure_class": "visible_answer_missing",
            "finish_reason": "tool_calls",
            "latency_ms": 797.134000051301,
            "reasoning_chars": 83,
            "reasoning_excerpt": "The user wants me to call the record_weather_zip function with the zip code 98101.\n",
            "reasoning_field": "reasoning_content",
            "reasoning_tokens": null,
            "status": "failed",
            "tool_call_count": 1,
            "usage": {
              "completion_tokens": 57,
              "prompt_tokens": 316,
              "total_tokens": 373
            },
            "valid_tool_call_count": 0,
            "validation_errors": [
              "missing required string argument: zip"
            ]
          }
        ],
        "error": "missing required string argument: zip",
        "expected_arguments": {
          "zip": "98101"
        },
        "expected_function": "record_weather_zip",
        "name": "openai_tool_call_smoke",
        "pass_count": 0,
        "pass_rate": 0.0,
        "required_pass_rate": 1.0,
        "status": "failed"
      }
    ],
    "status": "failed"
  },
  "voice": {
    "status": "not_run",
    "stt_latency_ms": null,
    "total_turn_latency_ms": null,
    "tts_latency_ms": null
  }
}
