{
  "checkedAt": "2026-08-20T10:21:39+08:00",
  "operationId": "20260820021047-bd395d1fc5",
  "status": "pinned-source-zero-gpu-audit",
  "scope": {
    "authenticatedApiCalls": 0,
    "liveModelCalls": 0,
    "graderCalls": 0,
    "gpuRuns": 0,
    "credentialsUsed": false,
    "description": "Pinned public-source audit and deterministic arithmetic only; no GLM, MiniMax, NVIDIA GPU, or serving endpoint was used."
  },
  "sourceReceipts": [
    {
      "url": "https://github.com/sgl-project/sglang/blob/a49560ce501ccce5f20d0e200d888606a547dffe/docs/cookbook/autoregressive/GLM/GLM-5.2.mdx",
      "sha256": "ed999a20d87297acaf62a4f5c14fd0e3fec054077e71594edb630f54fcc92b11",
      "bytes": 20574,
      "role": "Pinned SGLang GLM-5.2 architecture, hardware, backend, memory, MTP, and PD guidance"
    },
    {
      "url": "https://github.com/sgl-project/sglang/blob/a49560ce501ccce5f20d0e200d888606a547dffe/docs/src/snippets/configs/zai-org/glm-5.2.jsx",
      "sha256": "8b1b1f015ce63a72234c87f0cf07e4f2b6b1a0675f711dc2242b65b72b358bbf",
      "bytes": 47388,
      "role": "Pinned SGLang deployment cells and verified flags"
    },
    {
      "url": "https://github.com/sgl-project/sglang/blob/a49560ce501ccce5f20d0e200d888606a547dffe/docs/src/snippets/configs/zai-org/glm-5.2-benchmarks.jsx",
      "sha256": "1002ea1b91b3b36352c9bc7450d0a39c38b89e2b97e55324d580f5e262b2798e",
      "bytes": 13437,
      "role": "Pinned SGLang project B200 and B300 NVFP4 benchmark cells"
    },
    {
      "url": "https://github.com/sgl-project/sglang/releases/tag/v0.5.17",
      "sha256": "8af76ecf70597605a19ce93bcbea3dc791a017c9ca0ad36fceffeb07bcae7b38",
      "bytes": 66301,
      "role": "Latest SGLang release snapshot and later DSA/NVFP4 fixes at audit time"
    },
    {
      "url": "https://huggingface.co/nvidia/GLM-5.2-NVFP4/blob/aec724e8c7b8ee9db3b48c01c320f63f9cdaf8aa/README.md",
      "sha256": "1cc0ef3c7cf9a7addd2231f1bc3a036ed00ee6ead8f19583ecf312b4e95994b4",
      "bytes": 12453,
      "role": "Pinned NVIDIA quantization scope, runtime recipe, test hardware, and publisher accuracy table"
    },
    {
      "url": "https://huggingface.co/nvidia/GLM-5.2-NVFP4/blob/aec724e8c7b8ee9db3b48c01c320f63f9cdaf8aa/config.json",
      "sha256": "d3783a603e5aa9cb58eff7a5d8ac9c42d83156b229efe185c72e9e2dc8444923",
      "bytes": 15517,
      "role": "Pinned NVIDIA model architecture and context configuration"
    },
    {
      "url": "https://huggingface.co/nvidia/GLM-5.2-NVFP4/blob/aec724e8c7b8ee9db3b48c01c320f63f9cdaf8aa/model.safetensors.index.json",
      "sha256": "2aa8397b501d9f6a232d153f328feb912f813c389061aac4cf72b04914fa5b74",
      "bytes": 22194492,
      "role": "Pinned NVFP4 tensor-byte total, tensor map, and shard map"
    },
    {
      "url": "https://huggingface.co/zai-org/GLM-5.2-FP8/blob/ba978f7d347eaf65d22f1a86833408afdb953541/model.safetensors.index.json",
      "sha256": "e0fe7f28c1f853d4824e4d796374e3dacf1fe470988773952c79b063768134bf",
      "bytes": 11359251,
      "role": "Pinned official FP8 tensor-byte control"
    },
    {
      "url": "https://z.ai/blog/glm-5.2",
      "sha256": "63e94b9e11a2db64243ff18418011577f2a98ca7851ce9dcb0e97cfe34cc245a",
      "bytes": 598,
      "role": "Fixed per-round Z.AI GLM-5.2 release check"
    },
    {
      "url": "https://www.zhipuai.cn/zh/research",
      "sha256": "188772a5cb6b65eb65f02024d89d153b692ef3d3cb2c98ddd233352c0dd6ac80",
      "bytes": 1236461,
      "role": "Fixed per-round Zhipu research-index check"
    }
  ],
  "revisions": {
    "sglangCommit": "a49560ce501ccce5f20d0e200d888606a547dffe",
    "sglangLatestRelease": "v0.5.17",
    "sglangLatestReleasePublishedAt": "2026-08-08T00:19:16Z",
    "nvfp4Checkpoint": "aec724e8c7b8ee9db3b48c01c320f63f9cdaf8aa",
    "fp8Checkpoint": "ba978f7d347eaf65d22f1a86833408afdb953541"
  },
  "architecture": {
    "modelType": "glm_moe_dsa",
    "class": "GlmMoeDsaForCausalLM",
    "contextTokens": 1048576,
    "transformerLayers": 78,
    "routedExperts": 256,
    "activeExpertsPerToken": 8,
    "headDim": 192
  },
  "quantization": {
    "format": "NVFP4",
    "modelOptVersion": "0.46.0",
    "publisherVersion": "1.0",
    "scope": "Weights and activations of linear operators inside transformer-block MoE experts are quantized; the shared expert is not quantized.",
    "publisherTestHardware": [
      "NVIDIA B200",
      "NVIDIA B300"
    ],
    "publisherRuntimes": [
      "SGLang",
      "vLLM"
    ]
  },
  "checkpointArithmetic": {
    "nvfp4": {
      "repository": "nvidia/GLM-5.2-NVFP4",
      "tensorBytes": 464795267072,
      "tensorCount": 232385,
      "shards": 47,
      "decimalGB": 464.795,
      "gibibytes": 432.874,
      "weightOnlyGiBPerRankAtTp8": 54.109,
      "weightOnlyGiBPerRankAtTp4": 108.219
    },
    "fp8": {
      "repository": "zai-org/GLM-5.2-FP8",
      "tensorBytes": 755617140416,
      "tensorCount": 118629,
      "shards": 141,
      "decimalGB": 755.617,
      "gibibytes": 703.723
    },
    "nvfp4AsPercentOfFp8Bytes": 61.51,
    "nvfp4ReductionFromFp8Percent": 38.5,
    "nvfp4ShardReductionFromFp8Percent": 66.7,
    "sglangGb300CommentAudit": {
      "commentValueDecimalGB": 381,
      "indexedTensorDecimalGB": 464.795,
      "differenceDecimalGB": 83.795,
      "indexedValueAboveCommentPercent": 22,
      "conclusion": "Use the pinned Hugging Face tensor index for download and weight-floor planning; do not use the approximate SGLang source comment as an artifact-size receipt."
    }
  },
  "publisherAccuracyDeltas": [
    {
      "benchmark": "GPQA Diamond",
      "fp8": 89.52,
      "nvfp4": 89.39,
      "pointChange": -0.13,
      "relativeChangePercent": -0.15
    },
    {
      "benchmark": "SciCode",
      "fp8": 49.85,
      "nvfp4": 49.04,
      "pointChange": -0.81,
      "relativeChangePercent": -1.62
    },
    {
      "benchmark": "IFBench",
      "fp8": 74.95,
      "nvfp4": 75.81,
      "pointChange": 0.86,
      "relativeChangePercent": 1.15
    },
    {
      "benchmark": "AA-LCR",
      "fp8": 69.38,
      "nvfp4": 70.13,
      "pointChange": 0.75,
      "relativeChangePercent": 1.08
    },
    {
      "benchmark": "tau2-Bench Telecom",
      "fp8": 97.9,
      "nvfp4": 98.25,
      "pointChange": 0.35,
      "relativeChangePercent": 0.36
    }
  ],
  "sglangRecipes": [
    {
      "hardware": "B200",
      "strategy": "low-latency",
      "gpus": 8,
      "tp": 8,
      "dp": 1,
      "mtp": "5-1-6",
      "chunkedPrefill": 8192,
      "memFractionStatic": 0.85,
      "maxRunningRequests": null,
      "verified": true
    },
    {
      "hardware": "B200",
      "strategy": "balanced",
      "gpus": 8,
      "tp": 8,
      "dp": 8,
      "mtp": "2-1-3",
      "chunkedPrefill": 32768,
      "memFractionStatic": 0.92,
      "maxRunningRequests": 256,
      "verified": true
    },
    {
      "hardware": "B200",
      "strategy": "high-throughput",
      "gpus": 8,
      "tp": 8,
      "dp": 8,
      "mtp": "off",
      "chunkedPrefill": 32768,
      "memFractionStatic": 0.92,
      "maxRunningRequests": 512,
      "verified": true
    },
    {
      "hardware": "B300",
      "strategy": "low-latency",
      "gpus": 8,
      "tp": 8,
      "dp": 1,
      "mtp": "5-1-6",
      "chunkedPrefill": 8192,
      "memFractionStatic": 0.85,
      "maxRunningRequests": 16,
      "verified": true
    },
    {
      "hardware": "B300",
      "strategy": "balanced",
      "gpus": 8,
      "tp": 8,
      "dp": 8,
      "mtp": "2-1-3",
      "chunkedPrefill": 8192,
      "memFractionStatic": 0.85,
      "maxRunningRequests": 256,
      "verified": true
    },
    {
      "hardware": "B300",
      "strategy": "high-throughput",
      "gpus": 8,
      "tp": 8,
      "dp": 8,
      "mtp": "off",
      "chunkedPrefill": 8192,
      "memFractionStatic": 0.85,
      "maxRunningRequests": 1024,
      "verified": true
    },
    {
      "hardware": "GB300",
      "strategy": "low-latency",
      "gpus": 4,
      "tp": 4,
      "dp": 1,
      "mtp": "5-1-6",
      "chunkedPrefill": 8192,
      "memFractionStatic": 0.85,
      "maxRunningRequests": 16,
      "verified": true
    },
    {
      "hardware": "GB300",
      "strategy": "balanced",
      "gpus": 4,
      "tp": 4,
      "dp": 4,
      "mtp": "2-1-3",
      "chunkedPrefill": 8192,
      "memFractionStatic": 0.92,
      "maxRunningRequests": 256,
      "verified": true
    },
    {
      "hardware": "GB300",
      "strategy": "high-throughput",
      "gpus": 4,
      "tp": 4,
      "dp": 4,
      "mtp": "off",
      "chunkedPrefill": 8192,
      "memFractionStatic": 0.92,
      "maxRunningRequests": 512,
      "verified": true
    }
  ],
  "sglangBenchmarkPairs": [
    {
      "strategy": "low-latency",
      "concurrency": 1,
      "inputTokens": 8192,
      "outputTokens": 1024,
      "b200TokensPerSecondPerGpu": 527,
      "b300TokensPerSecondPerGpu": 459,
      "b200RelativeLeadPercent": 14.8,
      "b200TtftMs": 295,
      "b300TtftMs": 196,
      "boundary": "SGLang project cells, not a general hardware ranking or an independent GLM52.ai run."
    },
    {
      "strategy": "low-latency",
      "concurrency": 16,
      "inputTokens": 8192,
      "outputTokens": 1024,
      "b200TokensPerSecondPerGpu": 2289,
      "b300TokensPerSecondPerGpu": 2016,
      "b200RelativeLeadPercent": 13.5,
      "b200TtftMs": 2491,
      "b300TtftMs": 274,
      "boundary": "SGLang project cells, not a general hardware ranking or an independent GLM52.ai run."
    },
    {
      "strategy": "balanced",
      "concurrency": 64,
      "inputTokens": 8192,
      "outputTokens": 1024,
      "b200TokensPerSecondPerGpu": 3770,
      "b300TokensPerSecondPerGpu": 1377,
      "b200RelativeLeadPercent": 173.8,
      "b200TtftMs": 5837,
      "b300TtftMs": 680,
      "boundary": "SGLang project cells, not a general hardware ranking or an independent GLM52.ai run."
    },
    {
      "strategy": "balanced",
      "concurrency": 256,
      "inputTokens": 8192,
      "outputTokens": 1024,
      "b200TokensPerSecondPerGpu": 5343,
      "b300TokensPerSecondPerGpu": 1845,
      "b200RelativeLeadPercent": 189.6,
      "b200TtftMs": 16736,
      "b300TtftMs": 3010,
      "boundary": "SGLang project cells, not a general hardware ranking or an independent GLM52.ai run."
    },
    {
      "strategy": "high-throughput",
      "concurrency": 1024,
      "inputTokens": 8192,
      "outputTokens": 1024,
      "b200TokensPerSecondPerGpu": 5305,
      "b300TokensPerSecondPerGpu": 3870,
      "b200RelativeLeadPercent": 37.1,
      "b200TtftMs": 130174,
      "b300TtftMs": 6370,
      "boundary": "SGLang project cells, not a general hardware ranking or an independent GLM52.ai run."
    }
  ],
  "upstreamBoundaries": {
    "supportedHardwareInCurrentConfig": [
      "H200",
      "B200",
      "GB300",
      "B300",
      "MI355X",
      "MI325X",
      "MI300X"
    ],
    "h20W4aFp8PullRequest": {
      "url": "https://github.com/sgl-project/sglang/pull/31006",
      "state": "open",
      "mergedAt": null,
      "updatedAt": "2026-07-24T09:12:09Z"
    },
    "issues": [
      {
        "number": 31093,
        "url": "https://github.com/sgl-project/sglang/issues/31093",
        "state": "open",
        "scope": "NVFP4 plus a non-cookbook 6-1-7 EAGLE shape hit CUDA graph capture on v0.5.15 and a July main revision"
      },
      {
        "number": 30989,
        "url": "https://github.com/sgl-project/sglang/issues/30989",
        "state": "closed",
        "scope": "NVFP4 long-context NaN/logit collapse regression on v0.5.15"
      },
      {
        "number": 30033,
        "url": "https://github.com/sgl-project/sglang/issues/30033",
        "state": "open",
        "scope": "PD disaggregation plus DP-attention resident-KV admission-control gap observed on FP8"
      },
      {
        "number": 29110,
        "url": "https://github.com/sgl-project/sglang/issues/29110",
        "state": "closed",
        "scope": "Older PD plus EAGLE plus grammar double-accept bug fixed on main"
      }
    ]
  },
  "invariants": {
    "zeroModelAndGpuCalls": true,
    "pinnedSglangCommit": true,
    "pinnedCheckpoints": true,
    "exactNvfp4TensorBytes": true,
    "exactFp8TensorBytes": true,
    "allBlackwellRecipesMarkedVerified": true,
    "h20AbsentFromCurrentSupportedList": true,
    "h20RecipeStillUnmerged": true,
    "accuracyRowsFinite": true,
    "benchmarkPairsComplete": true
  },
  "limitations": [
    "GLM52.ai did not possess B200, B300, or GB300 hardware for this audit and did not reproduce the serving or performance rows.",
    "NVIDIA accuracy rows and SGLang performance cells are publisher/project evidence; they are not independent benchmarks.",
    "Tensor bytes are a download and weight-floor measurement, not peak GPU memory, KV-cache capacity, or proof that a topology fits a target context and concurrency.",
    "The SGLang main branch and issue state can change after the pinned audit; deployment must freeze an image digest and rerun the canary suite.",
    "An open issue is a rollout signal, not proof that every current recipe or version is affected; closed issues still justify regression fixtures."
  ]
}
