{
  "schema": "anvil-serving.benchmark-decision-summary/v1",
  "campaign_id": "2026-09-08-glm53-linux-wsl-comparison",
  "complete": true,
  "identity": {
    "schema": "anvil-serving.llm-qualification-configuration/v1",
    "model": "ormandj/GLM-5.3-Flash-W4A16-NVFP4-K32-Experts-FP8-WO",
    "revision": "c3cbb9891b67c741bcbf6b176dd7af9265b069db",
    "image": "ghcr.io/ormandj/sglang-glm53-flash-sm120@sha256:0c0637959c3931829f05154087bbefd2c50003fb9b2010200ce0ec82f4d71a53",
    "served_model": "glm53-flash-ormandj-sglang-sm120-tp2-393k-c1-adaptive-mtp",
    "os": "Linux-7.0.0-31-generic-x86_64-with-glibc2.43",
    "repository_revision": "8abcc5dc75a1b2389210b553120abb76e5864d3a",
    "declared_serve_flags": [
      "--served-model-name glm53-flash-ormandj-sglang-sm120-tp2-393k-c1-adaptive-mtp",
      "--tp-size 2",
      "--quantization modelopt_mixed",
      "--enable-multimodal",
      "--image-processor-backend torchvision",
      "--mm-process-config '{\"image\":{\"max_image_tokens\":3072}}'",
      "--warmups serving_coverage",
      "--moe-runner-backend flashinfer_cutlass",
      "--disable-shared-experts-fusion",
      "--disable-custom-all-reduce",
      "--attention-backend dsa",
      "--linear-attn-backend triton",
      "--dsa-prefill-backend flashinfer_sparse_mla",
      "--dsa-decode-backend flashinfer_sparse_mla",
      "--kv-cache-dtype fp8_e4m3",
      "--mamba-ssm-dtype bfloat16",
      "--context-length 393216",
      "--max-total-tokens 393216",
      "--mem-fraction-static 0.99",
      "--chunked-prefill-size 4096",
      "--max-prefill-tokens 4096",
      "--max-running-requests 1",
      "--max-mamba-cache-size 5",
      "--cuda-graph-max-bs-decode 1",
      "--speculative-algorithm EAGLE",
      "--speculative-num-steps 5",
      "--speculative-eagle-topk 1",
      "--speculative-num-draft-tokens 6",
      "--speculative-adaptive",
      "--speculative-adaptive-config /tmp/glm53-adaptive.json",
      "--reasoning-parser glm45",
      "--tool-call-parser glm47",
      "--enable-metrics",
      "--enable-cache-report",
      "--host 0.0.0.0",
      "--port 8001"
    ],
    "transport_controls": {
      "CUDA_DEVICE_ORDER": "PCI_BUS_ID",
      "NCCL_CUMEM_ENABLE": "0",
      "NCCL_DEBUG": "INFO",
      "NCCL_P2P_DISABLE": "1"
    },
    "runtime_identity_verified": "managed recipe status + native models/model-info",
    "comparison_differences": [
      "Linux host/driver/runtime and NCCL_P2P_DISABLE=1",
      "host port and cache volume labels",
      "Benchmark runner revision differs; prompt bytes and actual prompt token counts verified against historical runner in prompt-equivalence.json"
    ],
    "promotion_authorized": false,
    "oci_build_revision": "a547c90c74f1363920287eb80adc88a16d1e7005",
    "oci_build_revision_source": "narrow read-only image metadata inspection, matches historical image metadata"
  },
  "workload": {
    "manifest": "workload-manifest.json",
    "plan": "run-plan.json"
  },
  "measurement": {
    "path": "warm online direct capacity; routed acceptance; isolated SWE worker",
    "warm_cold_state": "existing warm serve, cache not reset",
    "sample_count": "capacity n3 per context per OS; endurance n60 per OS; suite-specific counts in gates",
    "statistics": [
      "per-request median decode",
      "median effective prefill",
      "median TTFT",
      "median E2E",
      "deterministic pass counts"
    ]
  },
  "gates": [
    {
      "artifact": "direct-preflight-disabled.json",
      "passed": false,
      "checks_passed": 9,
      "checks": 10
    },
    {
      "artifact": "direct-preflight-thinking-enabled.json",
      "passed": true,
      "checks_passed": 5,
      "checks": 5
    },
    {
      "artifact": "routed-preflight-disabled.json",
      "passed": false,
      "checks_passed": 9,
      "checks": 10
    },
    {
      "artifact": "routed-thinking-enabled.json",
      "passed": false,
      "checks_passed": 5,
      "checks": 5
    },
    {
      "artifact": "coding-quality.json",
      "passed": true,
      "attempts": 15,
      "passed_attempts": 15
    },
    {
      "artifact": "image-corpus.json",
      "passed": true,
      "attempts": 12,
      "passed_attempts": 12
    },
    {
      "artifact": "agentic/artifact.json",
      "completeness": "completed",
      "summary": {
        "attempted": 30,
        "pass_rate": 1.0,
        "passed": 30,
        "required_pass_rate": 0.7
      }
    },
    {
      "artifact": "swe/artifact.json",
      "completeness": "completed",
      "summary": {
        "attempted": 1,
        "graded": 1,
        "resolve_rate": 1.0,
        "resolved": 1
      }
    },
    {
      "artifact": "controlled-scout-4k-r3.json",
      "requests": 3,
      "completed": 0,
      "failed": 3,
      "performance_eligible": false
    },
    {
      "artifact": "linux-unique-natural-4k-r10.json",
      "requests": 10,
      "completed": 1,
      "failed": 9,
      "performance_eligible": false
    },
    {
      "artifact": "endurance-4k-r60.json",
      "requests": 60,
      "completed": 60,
      "failed": 0,
      "performance_eligible": true
    },
    {
      "artifact": "context/artifact.json",
      "passed": false,
      "attempts": 150,
      "passed_attempts": 128,
      "empty_length_terminated": 9,
      "incorrect_visible_answers": 13
    }
  ],
  "headline_results": {
    "schema": "anvil-serving.os-comparison/v1",
    "rows": [
      {
        "context_target": 4096,
        "actual_prompt_tokens": 2969.0,
        "sample_count_per_os": 3,
        "metrics": {
          "decode_tok_s_p50": {
            "windows": 112.06564058179691,
            "linux": 149.01900227284116,
            "change_percent": 32.974747209937135
          },
          "effective_prefill_tok_s_p50": {
            "windows": 16729.465552552232,
            "linux": 23143.00551077051,
            "change_percent": 38.33678928995931
          },
          "ttft_p50_ms": {
            "windows": 177.4718000087887,
            "linux": 128.2897810015129,
            "change_percent": -27.712582508793083
          },
          "e2e_p50_ms": {
            "windows": 500.499099958688,
            "linux": 611.2129540015303,
            "change_percent": 22.120689937700334
          },
          "throughput_tok_s": {
            "windows": 56.996474897490415,
            "linux": 106.75009323273754,
            "change_percent": 87.29244821671911
          }
        },
        "windows_source": "windows-capacity-4k-r3.json",
        "linux_source": "linux-legacy-capacity-4k-r3.json",
        "limitations": [
          "Historical variable-output, shared-prefix workload; descriptive medians only",
          "Whole migrated stack comparison, not isolated OS causality"
        ],
        "median_output_tokens": {
          "windows": 36,
          "linux": 74
        }
      },
      {
        "context_target": 120000,
        "actual_prompt_tokens": 96278.0,
        "sample_count_per_os": 3,
        "metrics": {
          "decode_tok_s_p50": {
            "windows": 96.16500114296164,
            "linux": 125.02757587394592,
            "change_percent": 30.01359578634679
          },
          "effective_prefill_tok_s_p50": {
            "windows": 5748.638371968673,
            "linux": 5879.3029344226425,
            "change_percent": 2.272965422405271
          },
          "ttft_p50_ms": {
            "windows": 16747.966599999927,
            "linux": 16376.091381996957,
            "change_percent": -2.2204201076145624
          },
          "e2e_p50_ms": {
            "windows": 17233.929299982265,
            "linux": 16738.361063999037,
            "change_percent": -2.875538290526336
          },
          "throughput_tok_s": {
            "windows": 2.715436165889082,
            "linux": 2.6245311985354314,
            "change_percent": -3.3477114467128977
          }
        },
        "windows_source": "windows-capacity-120k-r3.json",
        "linux_source": "linux-legacy-capacity-120k-r3.json",
        "limitations": [
          "Historical variable-output, shared-prefix workload; descriptive medians only",
          "Whole migrated stack comparison, not isolated OS causality"
        ],
        "median_output_tokens": {
          "windows": 48,
          "linux": 46
        }
      },
      {
        "context_target": 262144,
        "actual_prompt_tokens": 216101.0,
        "sample_count_per_os": 3,
        "metrics": {
          "decode_tok_s_p50": {
            "windows": 102.41880883554646,
            "linux": 124.46274256962307,
            "change_percent": 21.52332563198669
          },
          "effective_prefill_tok_s_p50": {
            "windows": 5578.783369064824,
            "linux": 5458.399114754668,
            "change_percent": -2.1578944071874995
          },
          "ttft_p50_ms": {
            "windows": 38736.223999992944,
            "linux": 39590.5463229974,
            "change_percent": 2.2054868409595363
          },
          "e2e_p50_ms": {
            "windows": 39107.24919999484,
            "linux": 39967.26827999737,
            "change_percent": 2.1991295670129807
          },
          "throughput_tok_s": {
            "windows": 1.2242969681044664,
            "linux": 1.0338363102010584,
            "change_percent": -15.556736875554888
          }
        },
        "windows_source": "windows-capacity-262k-r3.json",
        "linux_source": "linux-legacy-capacity-262k-r3.json",
        "limitations": [
          "Historical variable-output, shared-prefix workload; descriptive medians only",
          "Whole migrated stack comparison, not isolated OS causality"
        ],
        "median_output_tokens": {
          "windows": 39,
          "linux": 48
        }
      },
      {
        "context_target": 380000,
        "actual_prompt_tokens": 304491.0,
        "sample_count_per_os": 3,
        "metrics": {
          "decode_tok_s_p50": {
            "windows": 99.79115138353907,
            "linux": 120.28614187074675,
            "change_percent": 20.53788357290003
          },
          "effective_prefill_tok_s_p50": {
            "windows": 5456.732792727141,
            "linux": 5425.364604799794,
            "change_percent": -0.5748529224145926
          },
          "ttft_p50_ms": {
            "windows": 55801.15680000745,
            "linux": 56123.41785300305,
            "change_percent": 0.5775167962029704
          },
          "e2e_p50_ms": {
            "windows": 56362.32839996228,
            "linux": 56462.859640003444,
            "change_percent": 0.1783660166197576
          },
          "throughput_tok_s": {
            "windows": 0.8353306078293593,
            "linux": 0.7631981612491053,
            "change_percent": -8.635197358288249
          }
        },
        "windows_source": "windows-capacity-380k-r3.json",
        "linux_source": "linux-legacy-capacity-380k-r3.json",
        "limitations": [
          "Historical variable-output, shared-prefix workload; descriptive medians only",
          "Whole migrated stack comparison, not isolated OS causality"
        ],
        "median_output_tokens": {
          "windows": 51,
          "linux": 48
        }
      }
    ],
    "endurance": {
      "sample_count_per_os": 60,
      "windows_source": "windows-endurance-4k-r60.json",
      "linux_source": "endurance-4k-r60.json",
      "decode_windows": 102.19275206464441,
      "decode_linux": 142.6429205479183,
      "decode_change_percent": 39.58222835381338,
      "limitations": [
        "Variable-output shared-prefix historical workload",
        "n60 p99 is descriptive only"
      ]
    }
  },
  "failures": [
    "Strict output0/3; unique natural first-position canary1/10; excluded populations",
    "Direct general image phrase assertion failed; separate corpus12/12",
    "Routed nominal380K rejected413 by secondary context estimate",
    "Routed reasoning evidence absent despite correct visible answers",
    "Context quality128/150;9 empty length-terminated and13 incorrect visible answers"
  ],
  "limitations": [
    "Historical n=3 output-uncontrolled cache-reusing baseline is descriptive only, not an exact controlled-output comparison.",
    "OS migration also changes driver and NCCL transport; no causal OS-only claim.",
    "No speculation A/B because speculation is unchanged and not the target comparison.",
    "C1 is the deployed engine ceiling; higher client concurrency measures queueing, not increased engine capacity.",
    "No video qualification: existing contract explicitly excludes video.",
    "SWE one-instance smoke only; new trajectory slower with22 versus11 requests",
    "No controlled-output finalist performance qualification",
    "Default thinking in native context adapter; no reasoning-off performance claim from it",
    "New strict diagnostic failures have no matched retained Windows pass and do not establish Linux-caused regressions."
  ],
  "decision": {
    "evidence_labels": [
      "functional",
      "capacity",
      "quality"
    ],
    "decision_labels": [
      "no-promotion"
    ],
    "promotion_authorized": false,
    "next_gate": "Separate scoped fixes and platform/P2P qualification before any managed optimization comparison"
  },
  "evidence": [
    "historical-comparison.json",
    "agentic/artifact.json",
    "swe/artifact.json",
    "restoration.json",
    "context/artifact.json",
    "context-summary.json",
    "post-state.json"
  ]
}
