{
  "context": {
    "cap_tokens": 261056,
    "max_model_len": 262144,
    "targets": [
      {
        "attempted_context_tokens": 32768,
        "chars_per_token": 3.0,
        "clamped_tokens": 32768,
        "e2e_ms": 1918.4704999988753,
        "estimated_prompt_tokens": 32768,
        "output_tokens": 64,
        "status": "passed",
        "target_tokens": 32768,
        "ttft_ms": 1571.2552000004507,
        "usage": {
          "completion_tokens": 64,
          "prompt_tokens": 31225,
          "total_tokens": 31289
        }
      },
      {
        "attempted_context_tokens": 128000,
        "chars_per_token": 3.148,
        "clamped_tokens": 128000,
        "e2e_ms": 13211.844100000235,
        "estimated_prompt_tokens": 127999,
        "output_tokens": 64,
        "status": "passed",
        "target_tokens": 128000,
        "ttft_ms": 12839.492200000677,
        "usage": {
          "completion_tokens": 64,
          "prompt_tokens": 136117,
          "total_tokens": 136181
        }
      },
      {
        "attempted_context_tokens": 240000,
        "chars_per_token": 2.961,
        "clamped_tokens": 240000,
        "e2e_ms": 35614.604399999735,
        "estimated_prompt_tokens": 239999,
        "output_tokens": 64,
        "status": "passed",
        "target_tokens": 240000,
        "ttft_ms": 35208.91830000073,
        "usage": {
          "completion_tokens": 64,
          "prompt_tokens": 241327,
          "total_tokens": 241391
        }
      }
    ]
  },
  "evaluation_protocol": {
    "budget_semantics": "visible_answer_tokens plus reasoning_headroom_tokens are sent as one max_tokens cap; the endpoint does not hard-partition the channels",
    "minimum_comparable_repetitions": 3,
    "minimum_pass_rate": 1.0,
    "reasoning_headroom_tokens": 0,
    "records_finish_reason": true,
    "records_full_visible_answer": true,
    "records_reasoning_channel_metadata": true,
    "repetitions": 3,
    "version": 3,
    "visible_answer_tokens": 256
  },
  "failures": [],
  "identity": {
    "base_url": "http://127.0.0.1:30010/v1",
    "candidate_id": "InternScience/Agents-A1-FP8@4d7d59380f327b76e73bc71f40e0c589ad0ca1d5",
    "config_id": "agents-a1-fp8-mm-262144-f25953cc",
    "engine": "vllm-f25953cc",
    "gpu": "rtx-pro-6000-blackwell-max-q-96gb",
    "model": "agents-a1-fp8-mm-262k",
    "started_at": "2026-07-29T14:45:34Z"
  },
  "intelligence": {
    "checks": [
      {
        "attempt_count": 3,
        "attempts": [
          {
            "attempt": 1,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 256,
              "reasoning_headroom_tokens": 0,
              "visible_answer_tokens": 256
            },
            "content": "```diff\n--- a/app.py\n+++ b/app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "content_excerpt": "```diff\n--- a/app.py\n+++ b/app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 586.1208999995142,
            "reasoning_chars": 0,
            "reasoning_excerpt": "",
            "reasoning_field": null,
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "diff_shape",
                "passed": true
              },
              {
                "name": "removes_old_timeout",
                "passed": true
              },
              {
                "name": "adds_new_timeout",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 45,
              "prompt_tokens": 51,
              "prompt_tokens_details": null,
              "total_tokens": 96
            }
          },
          {
            "attempt": 2,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 256,
              "reasoning_headroom_tokens": 0,
              "visible_answer_tokens": 256
            },
            "content": "```diff\n--- a/app.py\n+++ b/app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "content_excerpt": "```diff\n--- a/app.py\n+++ b/app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 314.240300000165,
            "reasoning_chars": 0,
            "reasoning_excerpt": "",
            "reasoning_field": null,
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "diff_shape",
                "passed": true
              },
              {
                "name": "removes_old_timeout",
                "passed": true
              },
              {
                "name": "adds_new_timeout",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 45,
              "prompt_tokens": 51,
              "prompt_tokens_details": null,
              "total_tokens": 96
            }
          },
          {
            "attempt": 3,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 256,
              "reasoning_headroom_tokens": 0,
              "visible_answer_tokens": 256
            },
            "content": "```diff\n--- a/app.py\n+++ b/app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "content_excerpt": "```diff\n--- a/app.py\n+++ b/app.py\n@@ -1,2 +1,2 @@\n-timeout = 30\n+timeout = 45\n retries = 2\n```",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 310.38080000143964,
            "reasoning_chars": 0,
            "reasoning_excerpt": "",
            "reasoning_field": null,
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "diff_shape",
                "passed": true
              },
              {
                "name": "removes_old_timeout",
                "passed": true
              },
              {
                "name": "adds_new_timeout",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 45,
              "prompt_tokens": 51,
              "prompt_tokens_details": null,
              "total_tokens": 96
            }
          }
        ],
        "id": "unified_diff_timeout_edit",
        "pass_count": 3,
        "pass_rate": 1.0,
        "required_pass_rate": 1.0,
        "status": "passed",
        "validator": "deterministic_text_checks"
      },
      {
        "attempt_count": 3,
        "attempts": [
          {
            "attempt": 1,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 256,
              "reasoning_headroom_tokens": 0,
              "visible_answer_tokens": 256
            },
            "content": "The system exceeds its 2500 ms turn timeout because the sequential sum of STT, LLM, and TTS (3000 ms) is too long, so a practical fix is to run the LLM generation in parallel with the STT or TTS processes.",
            "content_excerpt": "The system exceeds its 2500 ms turn timeout because the sequential sum of STT, LLM, and TTS (3000 ms) is too long, so a practical fix is to run the LLM generation in parallel with the STT or TTS proce",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 752.9298999997991,
            "reasoning_chars": 0,
            "reasoning_excerpt": "",
            "reasoning_field": null,
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "identifies_budget_overrun",
                "passed": true
              },
              {
                "name": "offers_latency_fix",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 60,
              "prompt_tokens": 79,
              "prompt_tokens_details": null,
              "total_tokens": 139
            }
          },
          {
            "attempt": 2,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 256,
              "reasoning_headroom_tokens": 0,
              "visible_answer_tokens": 256
            },
            "content": "The system exceeds its 2500 ms turn timeout because the sequential sum of STT, LLM, and TTS (3000 ms) is too long, so a practical fix is to run the LLM generation in parallel with the STT or TTS processes.",
            "content_excerpt": "The system exceeds its 2500 ms turn timeout because the sequential sum of STT, LLM, and TTS (3000 ms) is too long, so a practical fix is to run the LLM generation in parallel with the STT or TTS proce",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 397.10010000089824,
            "reasoning_chars": 0,
            "reasoning_excerpt": "",
            "reasoning_field": null,
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "identifies_budget_overrun",
                "passed": true
              },
              {
                "name": "offers_latency_fix",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 60,
              "prompt_tokens": 79,
              "prompt_tokens_details": null,
              "total_tokens": 139
            }
          },
          {
            "attempt": 3,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 256,
              "reasoning_headroom_tokens": 0,
              "visible_answer_tokens": 256
            },
            "content": "The system exceeds its 2500 ms turn timeout because the sequential sum of STT, LLM, and TTS (3000 ms) is too long, so a practical fix is to run the LLM generation in parallel with the STT or TTS processes.",
            "content_excerpt": "The system exceeds its 2500 ms turn timeout because the sequential sum of STT, LLM, and TTS (3000 ms) is too long, so a practical fix is to run the LLM generation in parallel with the STT or TTS proce",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 390.17100000091887,
            "reasoning_chars": 0,
            "reasoning_excerpt": "",
            "reasoning_field": null,
            "reasoning_tokens": null,
            "status": "passed",
            "text_checks": [
              {
                "name": "identifies_budget_overrun",
                "passed": true
              },
              {
                "name": "offers_latency_fix",
                "passed": true
              }
            ],
            "usage": {
              "completion_tokens": 60,
              "prompt_tokens": 79,
              "prompt_tokens_details": null,
              "total_tokens": 139
            }
          }
        ],
        "id": "parallel_timeout_triage",
        "pass_count": 3,
        "pass_rate": 1.0,
        "required_pass_rate": 1.0,
        "status": "passed",
        "validator": "deterministic_text_checks"
      }
    ],
    "status": "passed"
  },
  "run_id": "fast-bakeoff-20260729T144534Z",
  "schema": "anvil-serving.fast-tier-bakeoff/v1",
  "score_inputs": {
    "e2e_p50_ms": 13211.844100000235,
    "intelligence_pass_rate": 1.0,
    "operational_fit_notes": [
      "endpoint was already loaded; benchmark did not start or stop serves"
    ],
    "session_recall_passed": true,
    "thinking_mode": "disabled",
    "tool_call_passed": true,
    "ttft_p50_ms": 12839.492200000677,
    "usable_context_tokens": 240000,
    "voice_latency_ms": null
  },
  "selection": {
    "context_targets": [
      32768,
      128000,
      240000
    ],
    "endpoint_already_loaded": true,
    "requests_per_context": 1,
    "suites": [
      "chat",
      "context",
      "tool",
      "session",
      "intelligence"
    ]
  },
  "session": {
    "checks": [
      {
        "attempt_count": 3,
        "attempts": [
          {
            "attempt": 1,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 256,
              "reasoning_headroom_tokens": 0,
              "visible_answer_tokens": 256
            },
            "content": "RIVER-918",
            "content_excerpt": "RIVER-918",
            "expected": "RIVER-918",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 442.6544999987527,
            "reasoning_chars": 0,
            "reasoning_excerpt": "",
            "reasoning_field": null,
            "reasoning_tokens": null,
            "status": "passed",
            "usage": {
              "completion_tokens": 7,
              "prompt_tokens": 52,
              "prompt_tokens_details": null,
              "total_tokens": 59
            }
          },
          {
            "attempt": 2,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 256,
              "reasoning_headroom_tokens": 0,
              "visible_answer_tokens": 256
            },
            "content": "RIVER-918",
            "content_excerpt": "RIVER-918",
            "expected": "RIVER-918",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 116.70349999985774,
            "reasoning_chars": 0,
            "reasoning_excerpt": "",
            "reasoning_field": null,
            "reasoning_tokens": null,
            "status": "passed",
            "usage": {
              "completion_tokens": 7,
              "prompt_tokens": 52,
              "prompt_tokens_details": null,
              "total_tokens": 59
            }
          },
          {
            "attempt": 3,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 256,
              "reasoning_headroom_tokens": 0,
              "visible_answer_tokens": 256
            },
            "content": "RIVER-918",
            "content_excerpt": "RIVER-918",
            "expected": "RIVER-918",
            "failure_class": null,
            "finish_reason": "stop",
            "latency_ms": 107.46959999960382,
            "reasoning_chars": 0,
            "reasoning_excerpt": "",
            "reasoning_field": null,
            "reasoning_tokens": null,
            "status": "passed",
            "usage": {
              "completion_tokens": 7,
              "prompt_tokens": 52,
              "prompt_tokens_details": null,
              "total_tokens": 59
            }
          }
        ],
        "name": "single_request_multiturn_recall",
        "pass_count": 3,
        "pass_rate": 1.0,
        "required_pass_rate": 1.0,
        "status": "passed"
      }
    ],
    "status": "passed"
  },
  "source_recipe": {
    "ref": "configs/agents-a1-qwen-262k-head-to-head-recipes.toml",
    "serve_command": "anvil-serving models recipes load InternScience/Agents-A1-FP8 --container agents-a1-fp8-primary-qualification --registry configs/agents-a1-qwen-262k-head-to-head-recipes.toml --confirm"
  },
  "suites": {},
  "thinking": {
    "chat_template_kwargs": {
      "enable_thinking": false
    },
    "control_evidence": "docs/findings/2026-07-28-agents-a1-multimodal-qualification-evidence/fp8-thinking-disabled-control.json",
    "control_evidence_sha256": "1d47f77e055ed9c51f6dc65659d9e843890539a53e7a05f5e47fade517e3f31f",
    "control_mechanism": "chat_template_kwargs",
    "control_requested": {
      "enable_thinking": false
    },
    "control_status": "verified",
    "mode": "disabled",
    "reasoning_effort": null,
    "unsupported": false
  },
  "timing": {
    "chat": {
      "content_chunks": 186,
      "decode_tok_s_p50": 169.19478590031324,
      "decode_tok_s_p95": 181.44361726077696,
      "e2e_p50_ms": 13211.844100000235,
      "e2e_p95_ms": 35614.604399999735,
      "effective_prefill_tok_s_p50": 10601.431729519087,
      "effective_prefill_tok_s_p95": 19872.647040398686,
      "generation_p50_ms": 372.35189999955765,
      "generation_p95_ms": 405.6860999990022,
      "mean_inter_token_latency_ms_p50": 5.910347619040597,
      "mean_inter_token_latency_ms_p95": 6.439461904746066,
      "output_tokens": 192,
      "prompt_token_samples": 3,
      "prompt_tokens": 408669,
      "prompt_tokens_p50": 136117.0,
      "prompt_tokens_p95": 241327.0,
      "ttft_p50_ms": 12839.492200000677,
      "ttft_p95_ms": 35208.91830000073,
      "ttft_samples": 3
    },
    "wall_ms": 55213.77699999903
  },
  "tool": {
    "checks": [
      {
        "attempt_count": 3,
        "attempts": [
          {
            "arguments": {
              "zip": "98101"
            },
            "attempt": 1,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 256,
              "reasoning_headroom_tokens": 0,
              "visible_answer_tokens": 256
            },
            "content": "",
            "content_excerpt": "",
            "failure_class": null,
            "finish_reason": "tool_calls",
            "latency_ms": 555.7098000008409,
            "reasoning_chars": 0,
            "reasoning_excerpt": "",
            "reasoning_field": null,
            "reasoning_tokens": null,
            "status": "passed",
            "tool_call_count": 1,
            "usage": {
              "completion_tokens": 32,
              "prompt_tokens": 288,
              "prompt_tokens_details": null,
              "total_tokens": 320
            },
            "valid_tool_call_count": 1,
            "validation_errors": []
          },
          {
            "arguments": {
              "zip": "98101"
            },
            "attempt": 2,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 256,
              "reasoning_headroom_tokens": 0,
              "visible_answer_tokens": 256
            },
            "content": "",
            "content_excerpt": "",
            "failure_class": null,
            "finish_reason": "tool_calls",
            "latency_ms": 244.18680000053428,
            "reasoning_chars": 0,
            "reasoning_excerpt": "",
            "reasoning_field": null,
            "reasoning_tokens": null,
            "status": "passed",
            "tool_call_count": 1,
            "usage": {
              "completion_tokens": 32,
              "prompt_tokens": 288,
              "prompt_tokens_details": null,
              "total_tokens": 320
            },
            "valid_tool_call_count": 1,
            "validation_errors": []
          },
          {
            "arguments": {
              "zip": "98101"
            },
            "attempt": 3,
            "budget": {
              "legacy_max_tokens_as_visible": false,
              "max_completion_tokens": 256,
              "reasoning_headroom_tokens": 0,
              "visible_answer_tokens": 256
            },
            "content": "",
            "content_excerpt": "",
            "failure_class": null,
            "finish_reason": "tool_calls",
            "latency_ms": 242.17999999927997,
            "reasoning_chars": 0,
            "reasoning_excerpt": "",
            "reasoning_field": null,
            "reasoning_tokens": null,
            "status": "passed",
            "tool_call_count": 1,
            "usage": {
              "completion_tokens": 32,
              "prompt_tokens": 288,
              "prompt_tokens_details": null,
              "total_tokens": 320
            },
            "valid_tool_call_count": 1,
            "validation_errors": []
          }
        ],
        "expected_arguments": {
          "zip": "98101"
        },
        "expected_function": "record_weather_zip",
        "name": "openai_tool_call_smoke",
        "pass_count": 3,
        "pass_rate": 1.0,
        "required_pass_rate": 1.0,
        "status": "passed"
      }
    ],
    "status": "passed"
  },
  "voice": {
    "llm_latency_ms": null,
    "status": "not_run",
    "stt_latency_ms": null,
    "total_turn_latency_ms": null,
    "tts_latency_ms": null
  }
}
