{
  "schemaVersion": 1,
  "operationId": "20260820150811-b51258fc8f",
  "checkedAt": "2026-08-20T23:16:36+08:00",
  "status": "pinned-source-zero-gpu-audit",
  "scope": {
    "authenticatedApiCalls": 0,
    "liveModelCalls": 0,
    "graderCalls": 0,
    "gpuRuns": 0,
    "credentialsUsed": false,
    "description": "Pinned public-source audit and deterministic arithmetic only; no GLM, MiniMax, AMD GPU, or serving endpoint was used."
  },
  "sourceReceipts": [
    {
      "url": "https://github.com/sgl-project/sglang/blob/7f8f030000b628ea2cb033e7457a13dd0ac80f99/docs/cookbook/autoregressive/GLM/GLM-5.2.mdx",
      "sha256": "ed999a20d87297acaf62a4f5c14fd0e3fec054077e71594edb630f54fcc92b11",
      "bytes": 20574,
      "role": "Pinned SGLang GLM-5.2 AMD deployment guidance and correctness boundaries"
    },
    {
      "url": "https://github.com/sgl-project/sglang/blob/7f8f030000b628ea2cb033e7457a13dd0ac80f99/docs/src/snippets/configs/zai-org/glm-5.2.jsx",
      "sha256": "8b1b1f015ce63a72234c87f0cf07e4f2b6b1a0675f711dc2242b65b72b358bbf",
      "bytes": 47388,
      "role": "Pinned SGLang AMD recipe cells, images, flags, and verified states"
    },
    {
      "url": "https://github.com/sgl-project/sglang/blob/7f8f030000b628ea2cb033e7457a13dd0ac80f99/docs/src/snippets/configs/zai-org/glm-5.2-benchmarks.jsx",
      "sha256": "1002ea1b91b3b36352c9bc7450d0a39c38b89e2b97e55324d580f5e262b2798e",
      "bytes": 13437,
      "role": "Pinned SGLang project MI355X FP8 performance cells"
    },
    {
      "url": "https://huggingface.co/zai-org/GLM-5.2/blob/b4734de4facf877f85769a911abafc5283eab3d9/model.safetensors.index.json",
      "sha256": "5fd47a926aefce0f2c917f42523e5e0f3c87e23e389e767c3681536a62f5cf5e",
      "bytes": 5408032,
      "role": "Pinned official BF16 tensor-byte control"
    },
    {
      "url": "https://huggingface.co/zai-org/GLM-5.2-FP8/blob/ba978f7d347eaf65d22f1a86833408afdb953541/model.safetensors.index.json",
      "sha256": "e0fe7f28c1f853d4824e4d796374e3dacf1fe470988773952c79b063768134bf",
      "bytes": 11359251,
      "role": "Pinned official FP8 tensor-byte control"
    },
    {
      "url": "https://huggingface.co/amd/GLM-5.2-MXFP4/blob/386bd0e4ec821f7b07975701cec3c3b953a5576a/README.md",
      "sha256": "82d727e4513e2a15234cf6afb1630abdc8b25e86ebe59149d557e53a17c629b8",
      "bytes": 2996,
      "role": "Pinned AMD quantization scope, tested hardware, software stack, and publisher GSM8K result"
    },
    {
      "url": "https://huggingface.co/amd/GLM-5.2-MXFP4/blob/386bd0e4ec821f7b07975701cec3c3b953a5576a/config.json",
      "sha256": "8b46225f7afd7181735c2dd97b925bcb70dfebb72b1a105d4ac0be7583a5c726",
      "bytes": 77279,
      "role": "Pinned AMD Quark MXFP4 configuration and excluded-layer boundary"
    },
    {
      "url": "https://huggingface.co/amd/GLM-5.2-MXFP4/blob/386bd0e4ec821f7b07975701cec3c3b953a5576a/model.safetensors.index.json",
      "sha256": "fd42188894abe9196fb70a2113c4fd1d0569b29a307a89c0e18e200111456fc6",
      "bytes": 11476511,
      "role": "Pinned AMD MXFP4 tensor-byte total, tensor map, and shard map"
    },
    {
      "url": "https://rocm.docs.amd.com/projects/ai-ecosystem/en/latest/optimization/workload-optimization.html",
      "sha256": "5656cdb69a0b93a60a891033ba7c3a32b26d1ef1c1f3b952e64ea93dd064b25e",
      "bytes": 220952,
      "role": "AMD hardware memory, bandwidth, architecture, FP8 encoding, and MXFP4 capability matrix"
    },
    {
      "url": "https://instinct.docs.amd.com/projects/system-acceptance/en/latest/gpus/mi300x.html",
      "sha256": "c9b47780fa68809536d1b5234ba65e03c7d6a30624a2bee14176e998d3e1ff0f",
      "bytes": 57724,
      "role": "AMD MI300X eight-GPU node memory confirmation"
    },
    {
      "url": "https://www.amd.com/en/products/accelerators/instinct/mi300.html",
      "sha256": "a4a71f80727bfb15b34f6e9843306fea5fca6c160d5b2517f700a0a072640470",
      "bytes": 377671,
      "role": "AMD MI325X per-GPU and eight-GPU platform memory confirmation"
    },
    {
      "url": "https://www.amd.com/en/products/accelerators/instinct/mi350/mi355x.html",
      "sha256": "64f94e9277cb873c92dd0de688c4c97a46fc2e0d801e6d1da1977abf85d887f8",
      "bytes": 236406,
      "role": "AMD MI355X memory, bandwidth, and MXFP4 capability confirmation"
    },
    {
      "url": "https://api.github.com/repos/sgl-project/sglang/issues/28685",
      "sha256": "ff32680dce872fc2c90377ef58a22f7fbefa54313a07f7926106991a62f7170c",
      "bytes": 8946,
      "role": "Closed SGLang gfx950 block-FP8 wrong-output investigation"
    },
    {
      "url": "https://api.github.com/repos/sgl-project/sglang/issues/29785",
      "sha256": "f7f6da2641a0700845b8af80c35c0da0bc594caf236f359caa62bec02b78181f",
      "bytes": 3956,
      "role": "Open SGLang gfx950 MTP draft-kernel failure report"
    },
    {
      "url": "https://api.github.com/repos/ROCm/rocm-libraries/pulls/8639",
      "sha256": "f2cb38141d4faecf19fca21bfbc0d6709a8a51b3ebf05435968dbcf3669a6854",
      "bytes": 20377,
      "role": "Merged ROCm CK gfx950 bpreshuffle determinism fix"
    },
    {
      "url": "https://z.ai/blog/glm-5.2",
      "sha256": "63e94b9e11a2db64243ff18418011577f2a98ca7851ce9dcb0e97cfe34cc245a",
      "bytes": 598,
      "role": "Fixed per-round Z.AI GLM-5.2 release check"
    },
    {
      "url": "https://www.zhipuai.cn/zh/research",
      "sha256": "188772a5cb6b65eb65f02024d89d153b692ef3d3cb2c98ddd233352c0dd6ac80",
      "bytes": 1236461,
      "role": "Fixed per-round Zhipu research-index check"
    }
  ],
  "revisions": {
    "sglangCommit": "7f8f030000b628ea2cb033e7457a13dd0ac80f99",
    "sglangLatestRelease": "v0.5.17",
    "sglangLatestReleasePublishedAt": "2026-08-08T00:19:16Z",
    "bf16Checkpoint": "b4734de4facf877f85769a911abafc5283eab3d9",
    "fp8Checkpoint": "ba978f7d347eaf65d22f1a86833408afdb953541",
    "amdMxfp4Checkpoint": "386bd0e4ec821f7b07975701cec3c3b953a5576a",
    "rocmFixMergeCommit": "eaf0131b646eaa0231646aa271970692832bdd24"
  },
  "checkpointArithmetic": {
    "bf16": {
      "repository": "zai-org/GLM-5.2",
      "tensorBytes": 1506659919872,
      "decimalGB": 1506.66,
      "gibibytes": 1403.186,
      "tensorCount": 59585,
      "shards": 282,
      "officialRecipeTensorParallel": 8,
      "evenWeightOnlyGiBPerRank": 175.398
    },
    "fp8": {
      "repository": "zai-org/GLM-5.2-FP8",
      "tensorBytes": 755617140416,
      "decimalGB": 755.617,
      "gibibytes": 703.723,
      "tensorCount": 118629,
      "shards": 141,
      "officialRecipeTensorParallel": 8,
      "evenWeightOnlyGiBPerRank": 87.965
    },
    "mxfp4": {
      "repository": "amd/GLM-5.2-MXFP4",
      "tensorBytes": 438001945864,
      "decimalGB": 438.002,
      "gibibytes": 407.921,
      "tensorCount": 117410,
      "shards": 282,
      "officialRecipeTensorParallel": 4,
      "evenWeightOnlyGiBPerRank": 101.98
    },
    "reductions": {
      "fp8FromBf16Percent": 49.85,
      "mxfp4FromBf16Percent": 70.93,
      "mxfp4FromFp8Percent": 42.03
    }
  },
  "hardwareCapacity": [
    {
      "name": "MI300X",
      "architecture": "CDNA 3",
      "memoryPerGpuDecimalGB": 192,
      "memoryBandwidthTbPerSecond": 5.3,
      "gpusPerNode": 8,
      "fp8Encoding": "FNUZ",
      "mxfp4": false,
      "nodeDecimalGB": 1536,
      "indexedTensorShareOfMarketedNodeMemoryPercent": {
        "bf16": 98.09,
        "fp8": 49.19,
        "mxfp4": 28.52
      },
      "route": {
        "recommendedControl": "FP8 TP8 canary; SGLang lists the cells but marks them unverified",
        "bf16Boundary": "Omitted for single-node because indexed BF16 tensors alone consume 98.09% of marketed node HBM",
        "mxfp4Boundary": "Not an AMD-published MXFP4 target"
      }
    },
    {
      "name": "MI325X",
      "architecture": "CDNA 3",
      "memoryPerGpuDecimalGB": 256,
      "memoryBandwidthTbPerSecond": 6,
      "gpusPerNode": 8,
      "fp8Encoding": "FNUZ",
      "mxfp4": false,
      "nodeDecimalGB": 2048,
      "indexedTensorShareOfMarketedNodeMemoryPercent": {
        "bf16": 73.57,
        "fp8": 36.9,
        "mxfp4": 21.39
      },
      "route": {
        "recommendedControl": "FP8 TP8 canary; SGLang lists the cells but marks them unverified",
        "bf16Boundary": "Single-node weight capacity exists, but every BF16 recipe cell remains unverified",
        "mxfp4Boundary": "Not an AMD-published MXFP4 target"
      }
    },
    {
      "name": "MI355X",
      "architecture": "CDNA 4",
      "memoryPerGpuDecimalGB": 288,
      "memoryBandwidthTbPerSecond": 8,
      "gpusPerNode": 8,
      "fp8Encoding": "OCP",
      "mxfp4": true,
      "nodeDecimalGB": 2304,
      "indexedTensorShareOfMarketedNodeMemoryPercent": {
        "bf16": 65.39,
        "fp8": 32.8,
        "mxfp4": 19.01
      },
      "route": {
        "recommendedControl": "FP8 TP8 with the pinned gfx950 image; three profile cells are verified",
        "bf16Boundary": "Single-node weight capacity exists, but every BF16 recipe cell remains unverified",
        "mxfp4Boundary": "AMD publishes a TP4 artifact, but all GLM-5.2 SGLang cells remain unverified"
      }
    }
  ],
  "verificationMatrix": {
    "profileCount": 19,
    "verifiedCount": 3,
    "unverifiedCount": 16,
    "verifiedProfiles": [
      "MI355X FP8 low-latency",
      "MI355X FP8 balanced",
      "MI355X FP8 high-throughput"
    ],
    "unverifiedProfiles": [
      "MI325X FP8 low-latency",
      "MI325X FP8 balanced",
      "MI325X FP8 high-throughput",
      "MI300X FP8 low-latency",
      "MI300X FP8 balanced",
      "MI300X FP8 high-throughput",
      "MI355X BF16 low-latency",
      "MI355X BF16 balanced",
      "MI355X BF16 high-throughput",
      "MI325X BF16 low-latency",
      "MI325X BF16 balanced",
      "MI325X BF16 high-throughput",
      "MI355X MXFP4 low-latency",
      "MI355X MXFP4 balanced",
      "MI355X MXFP4 high-throughput",
      "MI355X MXFP4 mtp-314"
    ],
    "profiles": [
      {
        "hardware": "MI355X",
        "precision": "FP8",
        "strategy": "low-latency",
        "gpus": 8,
        "tp": 8,
        "chunkedPrefill": 131072,
        "memFractionStatic": 0.8,
        "cudaGraphMaxBatch": null,
        "maxRunningRequests": null,
        "watchdogSeconds": 1200,
        "verified": true
      },
      {
        "hardware": "MI355X",
        "precision": "FP8",
        "strategy": "balanced",
        "gpus": 8,
        "tp": 8,
        "chunkedPrefill": 32768,
        "memFractionStatic": 0.85,
        "cudaGraphMaxBatch": 128,
        "maxRunningRequests": 80,
        "watchdogSeconds": 1200,
        "verified": true
      },
      {
        "hardware": "MI355X",
        "precision": "FP8",
        "strategy": "high-throughput",
        "gpus": 8,
        "tp": 8,
        "chunkedPrefill": null,
        "memFractionStatic": 0.85,
        "cudaGraphMaxBatch": 256,
        "maxRunningRequests": 256,
        "watchdogSeconds": 1200,
        "verified": true
      },
      {
        "hardware": "MI325X",
        "precision": "FP8",
        "strategy": "low-latency",
        "gpus": 8,
        "tp": 8,
        "chunkedPrefill": 131072,
        "memFractionStatic": 0.8,
        "cudaGraphMaxBatch": null,
        "maxRunningRequests": null,
        "watchdogSeconds": 1200,
        "verified": false
      },
      {
        "hardware": "MI325X",
        "precision": "FP8",
        "strategy": "balanced",
        "gpus": 8,
        "tp": 8,
        "chunkedPrefill": 32768,
        "memFractionStatic": 0.85,
        "cudaGraphMaxBatch": 128,
        "maxRunningRequests": 80,
        "watchdogSeconds": 1200,
        "verified": false
      },
      {
        "hardware": "MI325X",
        "precision": "FP8",
        "strategy": "high-throughput",
        "gpus": 8,
        "tp": 8,
        "chunkedPrefill": null,
        "memFractionStatic": 0.85,
        "cudaGraphMaxBatch": 256,
        "maxRunningRequests": 256,
        "watchdogSeconds": 1200,
        "verified": false
      },
      {
        "hardware": "MI300X",
        "precision": "FP8",
        "strategy": "low-latency",
        "gpus": 8,
        "tp": 8,
        "chunkedPrefill": 131072,
        "memFractionStatic": 0.8,
        "cudaGraphMaxBatch": null,
        "maxRunningRequests": null,
        "watchdogSeconds": 1200,
        "verified": false
      },
      {
        "hardware": "MI300X",
        "precision": "FP8",
        "strategy": "balanced",
        "gpus": 8,
        "tp": 8,
        "chunkedPrefill": 32768,
        "memFractionStatic": 0.85,
        "cudaGraphMaxBatch": 128,
        "maxRunningRequests": 80,
        "watchdogSeconds": 1200,
        "verified": false
      },
      {
        "hardware": "MI300X",
        "precision": "FP8",
        "strategy": "high-throughput",
        "gpus": 8,
        "tp": 8,
        "chunkedPrefill": null,
        "memFractionStatic": 0.85,
        "cudaGraphMaxBatch": 256,
        "maxRunningRequests": 256,
        "watchdogSeconds": 1200,
        "verified": false
      },
      {
        "hardware": "MI355X",
        "precision": "BF16",
        "strategy": "low-latency",
        "gpus": 8,
        "tp": 8,
        "chunkedPrefill": 131072,
        "memFractionStatic": 0.8,
        "cudaGraphMaxBatch": null,
        "maxRunningRequests": null,
        "watchdogSeconds": 1200,
        "verified": false
      },
      {
        "hardware": "MI355X",
        "precision": "BF16",
        "strategy": "balanced",
        "gpus": 8,
        "tp": 8,
        "chunkedPrefill": 32768,
        "memFractionStatic": 0.85,
        "cudaGraphMaxBatch": 128,
        "maxRunningRequests": 80,
        "watchdogSeconds": 1200,
        "verified": false
      },
      {
        "hardware": "MI355X",
        "precision": "BF16",
        "strategy": "high-throughput",
        "gpus": 8,
        "tp": 8,
        "chunkedPrefill": null,
        "memFractionStatic": 0.85,
        "cudaGraphMaxBatch": 256,
        "maxRunningRequests": 256,
        "watchdogSeconds": 1200,
        "verified": false
      },
      {
        "hardware": "MI325X",
        "precision": "BF16",
        "strategy": "low-latency",
        "gpus": 8,
        "tp": 8,
        "chunkedPrefill": 131072,
        "memFractionStatic": 0.8,
        "cudaGraphMaxBatch": null,
        "maxRunningRequests": null,
        "watchdogSeconds": 1200,
        "verified": false
      },
      {
        "hardware": "MI325X",
        "precision": "BF16",
        "strategy": "balanced",
        "gpus": 8,
        "tp": 8,
        "chunkedPrefill": 32768,
        "memFractionStatic": 0.85,
        "cudaGraphMaxBatch": 128,
        "maxRunningRequests": 80,
        "watchdogSeconds": 1200,
        "verified": false
      },
      {
        "hardware": "MI325X",
        "precision": "BF16",
        "strategy": "high-throughput",
        "gpus": 8,
        "tp": 8,
        "chunkedPrefill": null,
        "memFractionStatic": 0.85,
        "cudaGraphMaxBatch": 256,
        "maxRunningRequests": 256,
        "watchdogSeconds": 1200,
        "verified": false
      },
      {
        "hardware": "MI355X",
        "precision": "MXFP4",
        "strategy": "low-latency",
        "gpus": 4,
        "tp": 4,
        "chunkedPrefill": 131072,
        "memFractionStatic": 0.8,
        "cudaGraphMaxBatch": null,
        "maxRunningRequests": null,
        "watchdogSeconds": 1200,
        "verified": false
      },
      {
        "hardware": "MI355X",
        "precision": "MXFP4",
        "strategy": "balanced",
        "gpus": 4,
        "tp": 4,
        "chunkedPrefill": 32768,
        "memFractionStatic": 0.85,
        "cudaGraphMaxBatch": 128,
        "maxRunningRequests": 80,
        "watchdogSeconds": 1200,
        "verified": false
      },
      {
        "hardware": "MI355X",
        "precision": "MXFP4",
        "strategy": "high-throughput",
        "gpus": 4,
        "tp": 4,
        "chunkedPrefill": null,
        "memFractionStatic": 0.85,
        "cudaGraphMaxBatch": 256,
        "maxRunningRequests": 256,
        "watchdogSeconds": 1200,
        "verified": false
      },
      {
        "hardware": "MI355X",
        "precision": "MXFP4",
        "strategy": "mtp-314",
        "gpus": 4,
        "tp": 4,
        "chunkedPrefill": 131072,
        "memFractionStatic": 0.8,
        "cudaGraphMaxBatch": 160,
        "maxRunningRequests": 160,
        "watchdogSeconds": 1800,
        "verified": false
      }
    ]
  },
  "mi355xFp8ProjectBenchmarks": [
    {
      "hardware": "MI355X",
      "precision": "FP8",
      "strategy": "low-latency",
      "concurrency": 1,
      "inputTokens": 8192,
      "outputTokens": 1024,
      "ttftMs": 634,
      "tpotMs": 13.56,
      "tokensPerSecondPerGpu": 81
    },
    {
      "hardware": "MI355X",
      "precision": "FP8",
      "strategy": "low-latency",
      "concurrency": 16,
      "inputTokens": 8192,
      "outputTokens": 1024,
      "ttftMs": 5411,
      "tpotMs": 23.6,
      "tokensPerSecondPerGpu": 621
    },
    {
      "hardware": "MI355X",
      "precision": "FP8",
      "strategy": "balanced",
      "concurrency": 64,
      "inputTokens": 8192,
      "outputTokens": 1024,
      "ttftMs": 19526,
      "tpotMs": 46.5,
      "tokensPerSecondPerGpu": 1098
    },
    {
      "hardware": "MI355X",
      "precision": "FP8",
      "strategy": "balanced",
      "concurrency": 256,
      "inputTokens": 8192,
      "outputTokens": 1024,
      "ttftMs": 117866,
      "tpotMs": 56.12,
      "tokensPerSecondPerGpu": 1044
    },
    {
      "hardware": "MI355X",
      "precision": "FP8",
      "strategy": "high-throughput",
      "concurrency": 1024,
      "inputTokens": 8192,
      "outputTokens": 1024,
      "ttftMs": 432058,
      "tpotMs": 106.44,
      "tokensPerSecondPerGpu": 1269
    }
  ],
  "publisherAccuracy": {
    "amdMxfp4Gsm8k": {
      "bf16": 94.09,
      "mxfp4": 93.93,
      "extractor": "flexible-extract",
      "publisherClaimedRecoveryPercent": 99.8,
      "pointChange": -0.16,
      "calculatedRecoveryPercent": 99.83
    },
    "sglangMi355xFp8RegressionRetest": {
      "gsm8kApproximateFraction": 0.96,
      "invalidPercent": 0,
      "needleInHaystackPassed": 15,
      "needleInHaystackTotal": 15,
      "maximumTestedTokensApproximate": 118000
    }
  },
  "runtime": {
    "commonDsaFlags": [
      "--dsa-prefill-backend tilelang",
      "--dsa-decode-backend tilelang"
    ],
    "dockerImages": {
      "MI355X_FP8_BF16": "lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260618",
      "MI355X_MXFP4": "lmsysorg/sglang-rocm:v0.5.16-rocm720-mi35x-20260728",
      "MI325X_MI300X": "lmsysorg/sglang-rocm:v0.5.13.post1-rocm700-mi30x-20260616"
    }
  },
  "upstreamBoundaries": {
    "amdContextParallelVerified": false,
    "amdBaseMtpVerified": false,
    "mi355xMxfp4Verified": false,
    "mi325xFp8Verified": false,
    "mi300xFp8Verified": false,
    "mi355xBf16Verified": false,
    "mi325xBf16Verified": false,
    "issues": [
      {
        "number": 28685,
        "url": "https://github.com/sgl-project/sglang/issues/28685",
        "state": "closed",
        "scope": "Earlier gfx950 ROCm 7.2 block-FP8 bpreshuffle output corruption; fixed in the pinned MI355X image and newer"
      },
      {
        "number": 29785,
        "url": "https://github.com/sgl-project/sglang/issues/29785",
        "state": "open",
        "scope": "gfx950 MTP/EAGLE DSA draft-kernel crash; keep speculative flags off outside a separately reproduced experiment"
      }
    ],
    "rocmFix": {
      "pullRequest": 8639,
      "url": "https://github.com/ROCm/rocm-libraries/pull/8639",
      "state": "merged",
      "mergedAt": "2026-07-06T19:12:20Z",
      "mergeCommit": "eaf0131b646eaa0231646aa271970692832bdd24"
    }
  },
  "invariants": {
    "allSourceReceiptsHashed": true,
    "noAuthenticatedOrModelCalls": true,
    "threeVerifiedProfilesAreMi355xFp8": true,
    "sixteenProfilesRemainUnverified": true,
    "mi300xSingleNodeBf16HasNoRuntimeMargin": true,
    "mxfp4OfficialRecipeIsTp4": true,
    "fiveMi355xFp8BenchmarkRows": true,
    "amdBaseMtpAndContextParallelRemainUnverified": true,
    "mxfp4PublisherRecoveryReproducesClaim": true
  },
  "limits": [
    "Even weight-only GiB per rank is arithmetic, not actual placement or peak accelerator memory.",
    "Marketed HBM capacity is expressed in decimal GB; tensor GiB uses binary units. Runtime buffers, KV cache, communication workspaces, graphs, allocator behavior, and redundancy are excluded.",
    "SGLang verified means the upstream project marked a recipe cell verified; it is not independent reproduction by GLM52.ai.",
    "SGLang project benchmark rows use one 8,192-input/1,024-output random workload and do not predict application latency, quality, price, or availability.",
    "Publisher accuracy rows are test priors, not equivalence certificates. GLM52.ai made no model or GPU calls in this audit.",
    "Mutable issue and documentation state can change after the pinned audit; production should freeze checkpoint revisions and image digests, then rerun gates."
  ]
}
