{
  "base_url": "http://127.0.0.1:39273/v1",
  "completed": 8,
  "concurrency": 4,
  "context_distribution": {
    "4096": 8
  },
  "context_seed": 0,
  "context_tokens": 4096,
  "engine": "sglang",
  "failed": 0,
  "failures": [],
  "finished_at": "2026-09-05T03:28:23Z",
  "gpu": "Primary Node / RTX PRO 6000 Blackwell Max-Q / TP=1",
  "manifest": null,
  "max_context_tokens": 262144,
  "max_tokens": 512,
  "measurement_protocol": "capacity-v3",
  "metrics": {
    "content_chunks": 116,
    "content_chunks_s": null,
    "decode_tok_s_max": 123.88091016295779,
    "decode_tok_s_mean": 96.62432992178584,
    "decode_tok_s_min": 82.2359675958268,
    "decode_tok_s_p25": 84.0925669498156,
    "decode_tok_s_p50": 88.62025463186512,
    "decode_tok_s_p75": 95.76853844436704,
    "decode_tok_s_p90": 123.88091016295779,
    "decode_tok_s_p95": 123.88091016295779,
    "decode_tok_s_p99": 123.88091016295779,
    "decode_tok_s_samples": 8,
    "e2e_max_ms": 884.057900053449,
    "e2e_mean_ms": 684.3785625242162,
    "e2e_min_ms": 565.6022001057863,
    "e2e_p25_ms": 566.3448000559583,
    "e2e_p50_ms": 702.553499955684,
    "e2e_p75_ms": 710.4213000275195,
    "e2e_p90_ms": 884.057900053449,
    "e2e_p95_ms": 884.057900053449,
    "e2e_p99_ms": 884.057900053449,
    "e2e_samples": 8,
    "effective_prefill_tok_s_max": 26692.00179214657,
    "effective_prefill_tok_s_mean": 22730.015532594432,
    "effective_prefill_tok_s_min": 20198.36098770254,
    "effective_prefill_tok_s_p25": 20210.28089642525,
    "effective_prefill_tok_s_p50": 21542.266487010864,
    "effective_prefill_tok_s_p75": 23846.075657181515,
    "effective_prefill_tok_s_p90": 26692.00179214657,
    "effective_prefill_tok_s_p95": 26692.00179214657,
    "effective_prefill_tok_s_p99": 26692.00179214657,
    "effective_prefill_tok_s_samples": 8,
    "generation_max_ms": 705.2875000517815,
    "generation_mean_ms": 523.7256000255002,
    "generation_min_ms": 387.23830005619675,
    "generation_p25_ms": 387.4689000658691,
    "generation_p50_ms": 541.6369000449777,
    "generation_p75_ms": 558.9225000003353,
    "generation_p90_ms": 705.2875000517815,
    "generation_p95_ms": 705.2875000517815,
    "generation_p99_ms": 705.2875000517815,
    "generation_samples": 8,
    "mean_inter_token_latency_ms_max": 12.160129311237613,
    "mean_inter_token_latency_ms_mean": 10.585935670340174,
    "mean_inter_token_latency_ms_min": 8.072268751372272,
    "mean_inter_token_latency_ms_p25": 8.239112767153122,
    "mean_inter_token_latency_ms_p50": 11.178450000006706,
    "mean_inter_token_latency_ms_p75": 11.419922448409608,
    "mean_inter_token_latency_ms_p90": 12.160129311237613,
    "mean_inter_token_latency_ms_p95": 12.160129311237613,
    "mean_inter_token_latency_ms_p99": 12.160129311237613,
    "mean_inter_token_latency_ms_samples": 8,
    "output_token_sources": {
      "usage": 8
    },
    "output_tokens": 402,
    "prefix_cache_hit_avg": null,
    "prompt_token_samples": 8,
    "prompt_tokens": 28904,
    "prompt_tokens_max": 3613.0,
    "prompt_tokens_mean": 3613.0,
    "prompt_tokens_min": 3613.0,
    "prompt_tokens_p25": 3613.0,
    "prompt_tokens_p50": 3613.0,
    "prompt_tokens_p75": 3613.0,
    "prompt_tokens_p90": 3613.0,
    "prompt_tokens_p95": 3613.0,
    "prompt_tokens_p99": 3613.0,
    "prompt_tokens_samples": 8,
    "reasoning_chunks": 0,
    "throughput_tok_s": 265.9452185918038,
    "time_to_first_output_max_ms": 178.87589999008924,
    "time_to_first_output_mean_ms": 160.65296249871608,
    "time_to_first_output_min_ms": 135.35889994818717,
    "time_to_first_output_p25_ms": 142.9772999836132,
    "time_to_first_output_p50_ms": 151.64709999226034,
    "time_to_first_output_p75_ms": 178.36390004958957,
    "time_to_first_output_p90_ms": 178.87589999008924,
    "time_to_first_output_p95_ms": 178.87589999008924,
    "time_to_first_output_p99_ms": 178.87589999008924,
    "time_to_first_output_samples": 8,
    "tpot_max_ms": 12.160129311237613,
    "tpot_mean_ms": 10.585935670340174,
    "tpot_min_ms": 8.072268751372272,
    "tpot_p25_ms": 8.239112767153122,
    "tpot_p50_ms": 11.178450000006706,
    "tpot_p75_ms": 11.419922448409608,
    "tpot_p90_ms": 12.160129311237613,
    "tpot_p95_ms": 12.160129311237613,
    "tpot_p99_ms": 12.160129311237613,
    "tpot_samples": 8,
    "ttft_max_ms": 178.87619999237359,
    "ttft_mean_ms": 160.65337500185706,
    "ttft_min_ms": 135.35929995123297,
    "ttft_p25_ms": 142.97759998589754,
    "ttft_p50_ms": 151.64729999378324,
    "ttft_p75_ms": 178.36440005339682,
    "ttft_p90_ms": 178.87619999237359,
    "ttft_p95_ms": 178.87619999237359,
    "ttft_p99_ms": 178.87619999237359,
    "ttft_samples": 8
  },
  "model": "qwen38-27b-inferact-nvfp4-pro6000-dflash2-k12-chunk8k",
  "request_timings": [
    {
      "content_chunks": 15,
      "decode_tok_s": 88.62025463186512,
      "e2e_ms": 709.3537000473589,
      "effective_prefill_tok_s": 21542.266487010864,
      "generation_ms": 541.6369000449777,
      "mean_inter_token_latency_ms": 11.284102084270367,
      "output_token_source": "usage",
      "output_tokens": 49,
      "planned_context_tokens": 4096,
      "prompt_tokens": 3613,
      "reasoning_chunks": 0,
      "request_index": 0,
      "time_to_first_output_ms": 167.7168000023812,
      "tpot_ms": 11.284102084270367,
      "ttft_ms": 167.7176000084728,
      "visible_generation_ms": 541.6361000388861
    },
    {
      "content_chunks": 13,
      "decode_tok_s": 123.88091016295779,
      "e2e_ms": 566.3448000559583,
      "effective_prefill_tok_s": 20198.36098770254,
      "generation_ms": 387.4689000658691,
      "mean_inter_token_latency_ms": 8.072268751372272,
      "output_token_source": "usage",
      "output_tokens": 49,
      "planned_context_tokens": 4096,
      "prompt_tokens": 3613,
      "reasoning_chunks": 0,
      "request_index": 1,
      "time_to_first_output_ms": 178.87589999008924,
      "tpot_ms": 8.072268751372272,
      "ttft_ms": 178.87619999237359,
      "visible_generation_ms": 387.46860006358474
    },
    {
      "content_chunks": 17,
      "decode_tok_s": 82.2359675958268,
      "e2e_ms": 884.057900053449,
      "effective_prefill_tok_s": 20210.28089642525,
      "generation_ms": 705.2875000517815,
      "mean_inter_token_latency_ms": 12.160129311237613,
      "output_token_source": "usage",
      "output_tokens": 59,
      "planned_context_tokens": 4096,
      "prompt_tokens": 3613,
      "reasoning_chunks": 0,
      "request_index": 2,
      "time_to_first_output_ms": 178.77040000166744,
      "tpot_ms": 12.160129311237613,
      "ttft_ms": 178.7707000039518,
      "visible_generation_ms": 705.2872000494972
    },
    {
      "content_chunks": 12,
      "decode_tok_s": 121.37229192768193,
      "e2e_ms": 565.6022001057863,
      "effective_prefill_tok_s": 20256.341103751918,
      "generation_ms": 387.23830005619675,
      "mean_inter_token_latency_ms": 8.239112767153122,
      "output_token_source": "usage",
      "output_tokens": 48,
      "planned_context_tokens": 4096,
      "prompt_tokens": 3613,
      "reasoning_chunks": 0,
      "request_index": 3,
      "time_to_first_output_ms": 178.36390004958957,
      "tpot_ms": 8.239112767153122,
      "ttft_ms": 178.36440005339682,
      "visible_generation_ms": 387.2378000523895
    },
    {
      "content_chunks": 13,
      "decode_tok_s": 89.457840756044,
      "e2e_ms": 710.5695999925956,
      "effective_prefill_tok_s": 23825.051716679038,
      "generation_ms": 558.9225000003353,
      "mean_inter_token_latency_ms": 11.178450000006706,
      "output_token_source": "usage",
      "output_tokens": 51,
      "planned_context_tokens": 4096,
      "prompt_tokens": 3613,
      "reasoning_chunks": 0,
      "request_index": 4,
      "time_to_first_output_ms": 151.64709999226034,
      "tpot_ms": 11.178450000006706,
      "ttft_ms": 151.64729999378324,
      "visible_generation_ms": 558.9222999988124
    },
    {
      "content_chunks": 14,
      "decode_tok_s": 84.0925669498156,
      "e2e_ms": 710.4213000275195,
      "effective_prefill_tok_s": 23846.075657181515,
      "generation_ms": 558.907900005579,
      "mean_inter_token_latency_ms": 11.891657446927212,
      "output_token_source": "usage",
      "output_tokens": 48,
      "planned_context_tokens": 4096,
      "prompt_tokens": 3613,
      "reasoning_chunks": 0,
      "request_index": 5,
      "time_to_first_output_ms": 151.51340002194047,
      "tpot_ms": 11.891657446927212,
      "ttft_ms": 151.51390002574772,
      "visible_generation_ms": 558.9074000017717
    },
    {
      "content_chunks": 16,
      "decode_tok_s": 87.5662689057284,
      "e2e_ms": 702.553499955684,
      "effective_prefill_tok_s": 25269.74561985777,
      "generation_ms": 559.5761999720708,
      "mean_inter_token_latency_ms": 11.419922448409608,
      "output_token_source": "usage",
      "output_tokens": 50,
      "planned_context_tokens": 4096,
      "prompt_tokens": 3613,
      "reasoning_chunks": 0,
      "request_index": 6,
      "time_to_first_output_ms": 142.9772999836132,
      "tpot_ms": 11.419922448409608,
      "ttft_ms": 142.97759998589754,
      "visible_generation_ms": 559.5758999697864
    },
    {
      "content_chunks": 16,
      "decode_tok_s": 95.76853844436704,
      "e2e_ms": 626.1254999553785,
      "effective_prefill_tok_s": 26692.00179214657,
      "generation_ms": 490.7666000071913,
      "mean_inter_token_latency_ms": 10.441842553344495,
      "output_token_source": "usage",
      "output_tokens": 48,
      "planned_context_tokens": 4096,
      "prompt_tokens": 3613,
      "reasoning_chunks": 0,
      "request_index": 7,
      "time_to_first_output_ms": 135.35889994818717,
      "tpot_ms": 10.441842553344495,
      "ttft_ms": 135.35929995123297,
      "visible_generation_ms": 490.7662000041455
    }
  ],
  "requests": 8,
  "run_id": "benchmark-20260905T032821Z",
  "schema": "anvil-serving.benchmark/v1",
  "serve_flags": {
    "no_thinking": false,
    "reasoning_effort": null,
    "shared_prefix_burst": false,
    "thinking_mode": "disabled"
  },
  "source_recipe": null,
  "started_at": "2026-09-05T03:28:21Z",
  "tier": null,
  "timing_methodology": {
    "clock": "client time.perf_counter",
    "decode": "usage completion tokens after the first token divided by client-observed generation time; completion tokens may include reasoning tokens",
    "distributions": "request-level mean, min, p25, p50, p75, p90, p95, p99, and max using the nearest-rank percentile method",
    "effective_prefill": "usage.prompt_tokens divided by client-observed time to first output; includes queueing, scheduling, prefill, and first-token work",
    "generation": "client-observed E2E minus time to first output",
    "mean_inter_token_latency": "client-observed generation time divided by completion tokens after the first token; not a raw per-token timestamp trace",
    "time_to_first_output": "request start through the first non-empty streamed reasoning or content delta",
    "tpot": "alias of mean_inter_token_latency for capacity-v3: client-observed generation time divided by completion tokens after the first token",
    "ttft": "request start through the first non-empty streamed content delta"
  },
  "wall_clock_ms": 1511.589500005357
}
