{
  "base_url": "http://127.0.0.1:8001/v1",
  "completed": 5,
  "concurrency": 1,
  "context_distribution": {
    "4096": 5
  },
  "context_seed": 0,
  "context_tokens": 4096,
  "engine": "vllm-glm53-exl3-k3-b12x-dflash2-k5",
  "failed": 0,
  "failures": [],
  "finished_at": "2026-08-29T23:26:19Z",
  "gpu": "dual-rtx-pro-6000-max-q-tp2",
  "manifest": null,
  "max_context_tokens": 1048576,
  "max_tokens": 256,
  "measurement_protocol": "capacity-v4-reasoning",
  "metrics": {
    "content_chunks": 54,
    "content_chunks_s": null,
    "decode_tok_s_p50": 82.08867558287247,
    "decode_tok_s_p95": 85.74366312772678,
    "e2e_p50_ms": 1376.9880999971065,
    "e2e_p95_ms": 8173.586000004434,
    "effective_prefill_tok_s_p50": 3088.388149635691,
    "effective_prefill_tok_s_p95": 3204.608070927221,
    "generation_p50_ms": 419.8560999939218,
    "generation_p95_ms": 505.65690000075847,
    "mean_inter_token_latency_ms_p50": 12.181948276049987,
    "mean_inter_token_latency_ms_p95": 15.322936363659348,
    "output_token_sources": {
      "usage": 5
    },
    "output_tokens": 168,
    "prefix_cache_hit_avg": null,
    "prompt_token_samples": 5,
    "prompt_tokens": 14840,
    "prompt_tokens_p50": 2968.0,
    "prompt_tokens_p95": 2968.0,
    "reasoning_chunks": 9,
    "throughput_tok_s": 12.312425600980363,
    "time_to_first_output_p50_ms": 961.0190999956103,
    "time_to_first_output_p95_ms": 7738.287500003935,
    "ttft_p50_ms": 996.8640000006417,
    "ttft_p95_ms": 7813.805100005993,
    "ttft_samples": 5
  },
  "model": "glm53-flash-exl3-k3-dflash2-k5-fp8-tp2-1m-vision",
  "request_timings": [
    {
      "content_chunks": 11,
      "decode_tok_s": 82.70187009594271,
      "e2e_ms": 8173.586000004434,
      "effective_prefill_tok_s": 383.54739339918433,
      "generation_ms": 435.2985000004992,
      "mean_inter_token_latency_ms": 12.091625000013867,
      "output_token_source": "usage",
      "output_tokens": 37,
      "planned_context_tokens": 4096,
      "prompt_tokens": 2968,
      "reasoning_chunks": 3,
      "request_index": 0,
      "time_to_first_output_ms": 7738.287500003935,
      "ttft_ms": 7813.805100005993,
      "visible_generation_ms": 359.7808999984409
    },
    {
      "content_chunks": 12,
      "decode_tok_s": 65.26164282530408,
      "e2e_ms": 1431.8232000005082,
      "effective_prefill_tok_s": 3204.608070927221,
      "generation_ms": 505.65690000075847,
      "mean_inter_token_latency_ms": 15.322936363659348,
      "output_token_source": "usage",
      "output_tokens": 34,
      "planned_context_tokens": 4096,
      "prompt_tokens": 2968,
      "reasoning_chunks": 3,
      "request_index": 1,
      "time_to_first_output_ms": 926.1662999997498,
      "ttft_ms": 1003.229599999031,
      "visible_generation_ms": 428.59360000147717
    },
    {
      "content_chunks": 10,
      "decode_tok_s": 85.74366312772678,
      "e2e_ms": 1346.6970000008587,
      "effective_prefill_tok_s": 3202.2756008909255,
      "generation_ms": 419.8560999939218,
      "mean_inter_token_latency_ms": 11.662669444275606,
      "output_token_source": "usage",
      "output_tokens": 37,
      "planned_context_tokens": 4096,
      "prompt_tokens": 2968,
      "reasoning_chunks": 3,
      "request_index": 2,
      "time_to_first_output_ms": 926.8409000069369,
      "ttft_ms": 996.8640000006417,
      "visible_generation_ms": 349.833000000217
    },
    {
      "content_chunks": 11,
      "decode_tok_s": 71.19094097778894,
      "e2e_ms": 1376.9880999971065,
      "effective_prefill_tok_s": 3060.95224285444,
      "generation_ms": 407.3551999972551,
      "mean_inter_token_latency_ms": 14.046731034388108,
      "output_token_source": "usage",
      "output_tokens": 30,
      "planned_context_tokens": 4096,
      "prompt_tokens": 2968,
      "reasoning_chunks": 0,
      "request_index": 3,
      "time_to_first_output_ms": 969.6328999998514,
      "ttft_ms": 969.6342999959597,
      "visible_generation_ms": 407.35380000114674
    },
    {
      "content_chunks": 10,
      "decode_tok_s": 82.08867558287247,
      "e2e_ms": 1314.29560000106,
      "effective_prefill_tok_s": 3088.388149635691,
      "generation_ms": 353.27650000544963,
      "mean_inter_token_latency_ms": 12.181948276049987,
      "output_token_source": "usage",
      "output_tokens": 30,
      "planned_context_tokens": 4096,
      "prompt_tokens": 2968,
      "reasoning_chunks": 0,
      "request_index": 4,
      "time_to_first_output_ms": 961.0190999956103,
      "ttft_ms": 961.0198999944259,
      "visible_generation_ms": 353.275700006634
    }
  ],
  "requests": 5,
  "run_id": "benchmark-20260829T232606Z",
  "schema": "anvil-serving.benchmark/v1",
  "serve_flags": {
    "no_thinking": false,
    "reasoning_effort": "low",
    "shared_prefix_burst": false,
    "thinking_mode": "default"
  },
  "source_recipe": null,
  "started_at": "2026-08-29T23:26:06Z",
  "tier": null,
  "timing_methodology": {
    "clock": "client time.perf_counter",
    "decode": "usage completion tokens after the first token divided by client-observed generation time; completion tokens may include reasoning tokens",
    "effective_prefill": "usage.prompt_tokens divided by client-observed time to first output; includes queueing, scheduling, prefill, and first-token work",
    "generation": "client-observed E2E minus time to first output",
    "mean_inter_token_latency": "client-observed generation time divided by completion tokens after the first token; not a raw per-token timestamp trace",
    "time_to_first_output": "request start through the first non-empty streamed reasoning or content delta",
    "ttft": "request start through the first non-empty streamed content delta"
  },
  "wall_clock_ms": 13644.752499996684
}
