{
  "schema": "anvil-serving.benchmark-decision-summary/v1",
  "campaign_id": "2026-09-09-glm53-native-nccl-p2p",
  "complete": true,
  "identity": {"configuration": "configuration-and-identity.json"},
  "workload": {"manifest": "workload-manifest.json", "plan": "run-plan.json"},
  "measurement": {"path": "warm direct capacity plus direct/routed gates", "warm_cold_state": "unique-prefix policy; cold cache not established", "sample_count": "4K n=12 and 120K n=3 per arm", "statistics": ["per-arm medians"]},
  "gates": [{"baseline_preflight": "pass", "candidate_preflight": "pass"}, {"high_context": "380K target needle and long tool pass in both arms"}, {"routed_candidate": "smoke and JSON pass after readmission"}, {"quality": "raw candidate marker result failed; independently adjudicated diagnostic keyword false negative"}],
  "headline_results": [{"context": "4K; 2,970 actual prompt tokens p50; n=12 per arm", "ttft_p50_change_percent": -8.99345124044093, "e2e_p50_change_percent": -7.433052828748899, "decode_p50_change_percent": 2.1319498071867393}, {"context": "120K; 97,159 actual prompt tokens p50; n=3 per arm", "ttft_p50_change_percent": -8.932499106231962, "e2e_p50_change_percent": -8.858094418021745, "decode_p50_change_percent": 1.3811107884631024}],
  "failures": ["128-word scout failed before comparison", "both 380K strict capacity cells failed 33 versus 32 code words and are performance-ineligible", "candidate diagnostic marker failure retained unchanged"],
  "limitations": ["fixed 32-word output", "small n", "no p99 or service-tail claim", "partial cache metadata", "no high-context performance headline", "diagnostic quality is not executed-code or leaderboard evidence"],
  "decision": {"evidence_labels": ["functional", "capacity", "compatibility-only"], "decision_labels": ["current"], "promotion_authorized": true, "next_gate": "broader repeated controlled-output and quality populations before broader transport claims"},
  "evidence": ["comparison.json", "restoration.json", "diagnostic-adjudication.json"]
}
