{
  "image_digest": "sha256:0c0637959c3931829f05154087bbefd2c50003fb9b2010200ce0ec82f4d71a53",
  "model": "ormandj/GLM-5.3-Flash-W4A16-NVFP4-K32-Experts-FP8-WO",
  "revision": "c3cbb9891b67c741bcbf6b176dd7af9265b069db",
  "recipe_digest": "595966873909cb2b2679dd481741d63de77e16f76cc3e8859a12c9be75eea2bb",
  "registry_digest": "34360ae2272eab7309e53b8221df7fb4043c41fbe24a3974acf131557e0324db",
  "running": true,
  "served_identity": "glm53-flash-ormandj-sglang-sm120-tp2-393k-c1-adaptive-mtp",
  "native_kv_offload": false,
  "repository_revision": "730cd4ddbe598715b89b08a1e88334e46c0666ad",
  "deployed_router_revision": "81dfc4fa12f79e8ecc169a1bb4b443d50d15ccf3",
  "router_python_files_hash_matched": 286,
  "hardware": "2x RTX PRO 6000 Blackwell Max-Q 96 GB, native Linux, TP2 PCIe without NVLink",
  "engine": "SGLang v0.1.1-rc.14+a547c90c74f1363920287eb80adc88a16d1e7005",
  "quantization": "W4A16 NVFP4 K32 experts FP8 WO",
  "kv": "fp8_e4m3",
  "recurrent_state": "bfloat16",
  "context": 393216,
  "concurrency": 1,
  "speculation": "adaptive EAGLE [3,5]",
  "prefill_unchanged": true,
  "transport_unchanged": true,
  "reasoning_effort": "max",
  "upstream_timeout_seconds": 1200,
  "measurement_path": "real Pi -> temporary authenticated Anvil routes -> same running GLM backend",
  "cache_state": "warm/cache-uncontrolled",
  "production_configuration_changed": false
}
