{
  "schemaVersion": 1,
  "operationId": "20260830142315-959cb782e9",
  "checkedAt": "2026-08-30T14:42:46.434Z",
  "status": "pinned-zero-runtime-glm52-sglang-kv-slot-zero-audit",
  "sourceReceipts": [
    {
      "id": "zai-glm52-release",
      "url": "https://z.ai/blog/glm-5.2",
      "finalUrl": "https://z.ai/blog/glm-5.2",
      "purpose": "mandatory first-party GLM-5.2 release check",
      "httpStatus": 200,
      "bytes": 598,
      "sha256": "a9e8c2b6f34717d69e3a0aa26bb117256a4d8c95bd299910c2a693000ee88fe8"
    },
    {
      "id": "zhipu-research-index",
      "url": "http://zhipuai.cn/zh/research",
      "finalUrl": "https://www.zhipuai.cn/zh/research",
      "purpose": "mandatory Zhipu AI research-discovery check",
      "httpStatus": 200,
      "bytes": 1235119,
      "sha256": "7d50f4c290fbc240f50fabc8d18f7a499688f968f53951655572909ddf547d4e"
    },
    {
      "id": "amd-glm52-mxfp4-config",
      "url": "https://huggingface.co/amd/GLM-5.2-MXFP4/resolve/386bd0e4ec821f7b07975701cec3c3b953a5576a/config.json",
      "finalUrl": "https://huggingface.co/api/resolve-cache/models/amd/GLM-5.2-MXFP4/386bd0e4ec821f7b07975701cec3c3b953a5576a/config.json?%2Famd%2FGLM-5.2-MXFP4%2Fresolve%2F386bd0e4ec821f7b07975701cec3c3b953a5576a%2Fconfig.json=&etag=%22e990c0eb20d48f0232243121a107a8d574247b7f%22",
      "purpose": "pin the exact GLM-5.2 MXFP4 architecture used by the upstream incident",
      "httpStatus": 200,
      "bytes": 77279,
      "sha256": "8b46225f7afd7181735c2dd97b925bcb70dfebb72b1a105d4ac0be7583a5c726"
    },
    {
      "id": "zai-glm52-fp8-config",
      "url": "https://huggingface.co/zai-org/GLM-5.2-FP8/resolve/ba978f7d347eaf65d22f1a86833408afdb953541/config.json",
      "finalUrl": "https://huggingface.co/api/resolve-cache/models/zai-org/GLM-5.2-FP8/ba978f7d347eaf65d22f1a86833408afdb953541/config.json?%2Fzai-org%2FGLM-5.2-FP8%2Fresolve%2Fba978f7d347eaf65d22f1a86833408afdb953541%2Fconfig.json=&etag=%224e1f0168afd127189fb1c4ddb1d4476a4fca96ac%22",
      "purpose": "cross-check the official GLM-5.2 DSA and MTP fields",
      "httpStatus": 200,
      "bytes": 29464,
      "sha256": "22e49334abf8562fecf70ca3292ba3f5b33f5602fb2bf10b52dd64a66cfe65ff"
    },
    {
      "id": "sglang-issue-36207",
      "url": "https://api.github.com/repos/sgl-project/sglang/issues/36207",
      "finalUrl": "https://api.github.com/repos/sgl-project/sglang/issues/36207",
      "purpose": "capture the repeated-token incident, deterministic poison probe, and unresolved fused-writer boundary",
      "httpStatus": 200,
      "bytes": 7081,
      "sha256": "6133e06f37aca381ce29ae216ea4e58b854b7ee15b31c457e8bcac0f3a830954"
    },
    {
      "id": "sglang-issue-36207-comments",
      "url": "https://api.github.com/repos/sgl-project/sglang/issues/36207/comments",
      "finalUrl": "https://api.github.com/repos/sgl-project/sglang/issues/36207/comments",
      "purpose": "capture the linked AITER follow-up",
      "httpStatus": 200,
      "bytes": 1866,
      "sha256": "97c58c68a2781e8cffd9b71e4688c3264cb987b2064225d033dd13232b1eb9fe"
    },
    {
      "id": "sglang-pr-36003",
      "url": "https://api.github.com/repos/sgl-project/sglang/pulls/36003",
      "finalUrl": "https://api.github.com/repos/sgl-project/sglang/pulls/36003",
      "purpose": "capture the merged SGLang-owned MLA writer fix and its stated limits",
      "httpStatus": 200,
      "bytes": 29194,
      "sha256": "c8985b004cfd74c64f78402499f2a99073968880d60b1307474a000fbbd73047"
    },
    {
      "id": "sglang-pr-36003-files",
      "url": "https://api.github.com/repos/sgl-project/sglang/pulls/36003/files?per_page=100",
      "finalUrl": "https://api.github.com/repos/sgl-project/sglang/pulls/36003/files?per_page=100",
      "purpose": "capture the exact six-file fix surface",
      "httpStatus": 200,
      "bytes": 23369,
      "sha256": "d650112e85df65a69009e459945dc5b3e49c706506a794c0acd8b61a7f410bd6"
    },
    {
      "id": "sglang-pr-36003-reviews",
      "url": "https://api.github.com/repos/sgl-project/sglang/pulls/36003/reviews?per_page=100",
      "finalUrl": "https://api.github.com/repos/sgl-project/sglang/pulls/36003/reviews?per_page=100",
      "purpose": "capture review state for the merged writer fix",
      "httpStatus": 200,
      "bytes": 4774,
      "sha256": "ffe4284d9d70b17aba515ff1f134c27ae0eb381d47d7cf6246ba314e2678dc0b"
    },
    {
      "id": "sglang-v0518-commit",
      "url": "https://api.github.com/repos/sgl-project/sglang/commits/v0.5.18",
      "finalUrl": "https://api.github.com/repos/sgl-project/sglang/commits/v0.5.18",
      "purpose": "resolve the last tagged release named in the audit",
      "httpStatus": 200,
      "bytes": 12417,
      "sha256": "8241eec1d54eea1101e9c713c396de5c10fbed9fa30bc44895545deb6f1f4ff8"
    },
    {
      "id": "sglang-issue-runtime-commit",
      "url": "https://api.github.com/repos/sgl-project/sglang/commits/db570fe619ae15d8bd4ee04f146bc8234c52a20d",
      "finalUrl": "https://api.github.com/repos/sgl-project/sglang/commits/db570fe619ae15d8bd4ee04f146bc8234c52a20d",
      "purpose": "pin the reported runtime revision",
      "httpStatus": 200,
      "bytes": 10578,
      "sha256": "32436cf75574d5f12a3052c520c30b5f140c49ed00715653109200a141a95820"
    },
    {
      "id": "sglang-fix-merge-commit",
      "url": "https://api.github.com/repos/sgl-project/sglang/commits/04c1036bb395cce7d9b5eab1a814163ab1dcefed",
      "finalUrl": "https://api.github.com/repos/sgl-project/sglang/commits/04c1036bb395cce7d9b5eab1a814163ab1dcefed",
      "purpose": "pin the merged standard-writer fix",
      "httpStatus": 200,
      "bytes": 27930,
      "sha256": "23b394062d4fc20b725d415a798203a006c69a3b71ff51050ec3bc75ac252ea0"
    },
    {
      "id": "sglang-v0518-mla-buffer",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/71de97b264b04dcd514cf904003028aefe9775c8/python/sglang/kernels/ops/kvcache/mla_buffer.py",
      "finalUrl": "https://raw.githubusercontent.com/sgl-project/sglang/71de97b264b04dcd514cf904003028aefe9775c8/python/sglang/kernels/ops/kvcache/mla_buffer.py",
      "purpose": "verify v0.5.18 predates the reserved MLA writer guard",
      "httpStatus": 200,
      "bytes": 11578,
      "sha256": "3726d1c298e50cfcc3704573ffc001ac9343294480d331552e22f907bcf546de"
    },
    {
      "id": "sglang-merged-mla-buffer",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/04c1036bb395cce7d9b5eab1a814163ab1dcefed/python/sglang/kernels/ops/kvcache/mla_buffer.py",
      "finalUrl": "https://raw.githubusercontent.com/sgl-project/sglang/04c1036bb395cce7d9b5eab1a814163ab1dcefed/python/sglang/kernels/ops/kvcache/mla_buffer.py",
      "purpose": "pin BF16, FP8, and scale writer guards after merge",
      "httpStatus": 200,
      "bytes": 12513,
      "sha256": "e4c8c51848244835ecdf62c4c2c1e0ee34a6ae191db279c7fda9f9692545ff7b"
    },
    {
      "id": "sglang-merged-set-mla-buffer",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/04c1036bb395cce7d9b5eab1a814163ab1dcefed/python/sglang/kernels/ops/kvcache/set_mla_kv_buffer.py",
      "finalUrl": "https://raw.githubusercontent.com/sgl-project/sglang/04c1036bb395cce7d9b5eab1a814163ab1dcefed/python/sglang/kernels/ops/kvcache/set_mla_kv_buffer.py",
      "purpose": "pin the Python CUDA TMA/JIT reserved-slot argument",
      "httpStatus": 200,
      "bytes": 4276,
      "sha256": "e913f6066629f188614f3fd53a186f454a30231b8fdea8591732137b5e575235"
    },
    {
      "id": "sglang-merged-set-mla-cuh",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/04c1036bb395cce7d9b5eab1a814163ab1dcefed/python/sglang/kernels/jit/csrc/elementwise/set_mla_kv_buffer.cuh",
      "finalUrl": "https://raw.githubusercontent.com/sgl-project/sglang/04c1036bb395cce7d9b5eab1a814163ab1dcefed/python/sglang/kernels/jit/csrc/elementwise/set_mla_kv_buffer.cuh",
      "purpose": "pin the CUDA TMA/JIT write predicate",
      "httpStatus": 200,
      "bytes": 8117,
      "sha256": "d942626eedd63be6661c5664c1b16c9beef063060fff9f436aea75da3d1798ad"
    },
    {
      "id": "sglang-merged-mla-tests",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/04c1036bb395cce7d9b5eab1a814163ab1dcefed/test/registered/kernels/ops/kvcache/test_set_mla_kv_buffer.py",
      "finalUrl": "https://raw.githubusercontent.com/sgl-project/sglang/04c1036bb395cce7d9b5eab1a814163ab1dcefed/test/registered/kernels/ops/kvcache/test_set_mla_kv_buffer.py",
      "purpose": "pin deterministic sentinel, dtype, quantized, scale, and opt-out regression coverage",
      "httpStatus": 200,
      "bytes": 10926,
      "sha256": "a69b463d88ef1a8158e2b9c38a2effd606e1b24563f68e9216712408387a032d"
    },
    {
      "id": "sglang-memory-pool-at-incident",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/db570fe619ae15d8bd4ee04f146bc8234c52a20d/python/sglang/srt/mem_cache/memory_pool.py",
      "finalUrl": "https://raw.githubusercontent.com/sgl-project/sglang/db570fe619ae15d8bd4ee04f146bc8234c52a20d/python/sglang/srt/mem_cache/memory_pool.py",
      "purpose": "pin the runtime comments that reserve slot 0 for padded tokens",
      "httpStatus": 200,
      "bytes": 212659,
      "sha256": "78e0a9105888fd03065d4d9971929f0e15852cda99e97160ff5ca33f116272be"
    },
    {
      "id": "sglang-rocm-linear-utils-at-incident",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/db570fe619ae15d8bd4ee04f146bc8234c52a20d/python/sglang/srt/layers/rocm_linear_utils.py",
      "finalUrl": "https://raw.githubusercontent.com/sgl-project/sglang/db570fe619ae15d8bd4ee04f146bc8234c52a20d/python/sglang/srt/layers/rocm_linear_utils.py",
      "purpose": "pin SGLang use of the fused AITER MLA writer",
      "httpStatus": 200,
      "bytes": 839,
      "sha256": "87ca6a653584c01bc264be38697ad93af3a52fdde023be1edb2c59043c8c04b0"
    },
    {
      "id": "sglang-pr-32477",
      "url": "https://api.github.com/repos/sgl-project/sglang/pulls/32477",
      "finalUrl": "https://api.github.com/repos/sgl-project/sglang/pulls/32477",
      "purpose": "separate the earlier generic KV-store fix from MLA-specific writers",
      "httpStatus": 200,
      "bytes": 34257,
      "sha256": "6c38c2071bdd4819f57382e1b7a6eb4990f8f690ab07c91097832aa9c16db330"
    },
    {
      "id": "sglang-generic-store-after-fix",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/ee678910f7000aa43886f218de0e159bf418f1b5/python/sglang/kernels/ops/kvcache/kvcache.py",
      "finalUrl": "https://raw.githubusercontent.com/sgl-project/sglang/ee678910f7000aa43886f218de0e159bf418f1b5/python/sglang/kernels/ops/kvcache/kvcache.py",
      "purpose": "pin the generic store reserved-slot contract",
      "httpStatus": 200,
      "bytes": 2953,
      "sha256": "23fa2a12601ba506e517d81e8b03ff4f7f16c2b44608541fd66c659551a14ceb"
    },
    {
      "id": "aiter-pr-5010",
      "url": "https://api.github.com/repos/ROCm/aiter/pulls/5010",
      "finalUrl": "https://api.github.com/repos/ROCm/aiter/pulls/5010",
      "purpose": "capture the open fused-writer pad_slot_id proposal",
      "httpStatus": 200,
      "bytes": 17368,
      "sha256": "664c01f661f1ee2db86d244ac90c7bdf524d8cfd7e7b6d0e000596cf462e0f83"
    },
    {
      "id": "aiter-pr-5010-files",
      "url": "https://api.github.com/repos/ROCm/aiter/pulls/5010/files?per_page=100",
      "finalUrl": "https://api.github.com/repos/ROCm/aiter/pulls/5010/files?per_page=100",
      "purpose": "capture the proposed AITER implementation and test surface",
      "httpStatus": 200,
      "bytes": 9496,
      "sha256": "0dcb2dd4756d74551c1eb2bf2d1081220de37099cf6f5832706402e52bf968a8"
    },
    {
      "id": "aiter-pr-5010-reviews",
      "url": "https://api.github.com/repos/ROCm/aiter/pulls/5010/reviews?per_page=100",
      "finalUrl": "https://api.github.com/repos/ROCm/aiter/pulls/5010/reviews?per_page=100",
      "purpose": "capture approval state for the open AITER proposal",
      "httpStatus": 200,
      "bytes": 2,
      "sha256": "4f53cda18c2baa0c0354bb5f9a3ecbe5ed12ab4d8e11ba873c2f11161202b945"
    },
    {
      "id": "aiter-proposed-wrapper",
      "url": "https://raw.githubusercontent.com/ROCm/aiter/5448910ddc44a1539b001bf648a8ac75b8f85e98/aiter/ops/triton/fusions/fused_kv_cache.py",
      "finalUrl": "https://raw.githubusercontent.com/ROCm/aiter/5448910ddc44a1539b001bf648a8ac75b8f85e98/aiter/ops/triton/fusions/fused_kv_cache.py",
      "purpose": "pin the backward-compatible pad_slot_id wrapper proposal",
      "httpStatus": 200,
      "bytes": 28593,
      "sha256": "c80d386cd3114b4b3b5dd00e5fcfa3babe9a241797d7387eb28315418eedc7fc"
    },
    {
      "id": "aiter-proposed-triton-kernel",
      "url": "https://raw.githubusercontent.com/ROCm/aiter/5448910ddc44a1539b001bf648a8ac75b8f85e98/aiter/ops/triton/_triton_kernels/fusions/fused_kv_cache.py",
      "finalUrl": "https://raw.githubusercontent.com/ROCm/aiter/5448910ddc44a1539b001bf648a8ac75b8f85e98/aiter/ops/triton/_triton_kernels/fusions/fused_kv_cache.py",
      "purpose": "pin both proposed fused KV write guards",
      "httpStatus": 200,
      "bytes": 37290,
      "sha256": "0ccdf779b09c61b660cb66cb9c5b312bb40a6ca3509ee8abaa7797f76af6d303"
    },
    {
      "id": "aiter-proposed-tests",
      "url": "https://raw.githubusercontent.com/ROCm/aiter/5448910ddc44a1539b001bf648a8ac75b8f85e98/op_tests/triton_tests/fusions/test_fused_kv_cache.py",
      "finalUrl": "https://raw.githubusercontent.com/ROCm/aiter/5448910ddc44a1539b001bf648a8ac75b8f85e98/op_tests/triton_tests/fusions/test_fused_kv_cache.py",
      "purpose": "pin the proposed default, reserved-slot, and positive-slot tests",
      "httpStatus": 200,
      "bytes": 30678,
      "sha256": "1c0e96cd8c5085a759635c7a115bf5cde0dca79a02a3e7213e5639e7d7f818a4"
    }
  ],
  "pins": {
    "sglangReleaseTag": "v0.5.18",
    "sglangReleaseRevision": "71de97b264b04dcd514cf904003028aefe9775c8",
    "issueRuntimeRevision": "db570fe619ae15d8bd4ee04f146bc8234c52a20d",
    "sglangFixMergeRevision": "04c1036bb395cce7d9b5eab1a814163ab1dcefed",
    "sglangFixHeadRevision": "01950f8813aecb012fb5a0a1e386bcaabfa89b24",
    "genericFixMergeRevision": "ee678910f7000aa43886f218de0e159bf418f1b5",
    "aiterFixHeadRevision": "5448910ddc44a1539b001bf648a8ac75b8f85e98",
    "aiterFixBaseRevision": "b12a1904fd74588e127b2deb785193e59f0ad52d",
    "amdModelRevision": "386bd0e4ec821f7b07975701cec3c3b953a5576a",
    "zaiModelRevision": "ba978f7d347eaf65d22f1a86833408afdb953541"
  },
  "modelContract": {
    "amdArchitectures": [
      "GlmMoeDsaForCausalLM"
    ],
    "amdModelType": "glm_moe_dsa",
    "amdQuantizationMethod": "quark",
    "amdIndexTopK": 2048,
    "amdIndexShareForMtp": true,
    "amdNextPredictLayers": 1,
    "amdKvLoraRank": 512,
    "amdNopeHeadDim": 192,
    "amdRopeHeadDim": 64,
    "zaiArchitectures": [
      "GlmMoeDsaForCausalLM"
    ],
    "zaiIndexTopK": 2048,
    "zaiIndexShareForMtp": true
  },
  "issueStatus": {
    "number": 36207,
    "title": "[Bug] MLA KV writers can overwrite reserved padding slot 0",
    "url": "https://github.com/sgl-project/sglang/issues/36207",
    "state": "open",
    "createdAt": "2026-08-24T17:04:45Z",
    "updatedAt": "2026-08-26T07:36:15Z",
    "commentCount": 1,
    "environmentRevision": "db570fe619ae15d8bd4ee04f146bc8234c52a20d",
    "environmentImage": "lmsysorg/sglang-rocm:v0.5.18-rocm724-mi35x-20260822",
    "environmentGpuCount": 8,
    "environmentGpu": "AMD MI350X (gfx950)",
    "model": "amd/GLM-5.2-MXFP4",
    "kvCacheDtype": "BF16",
    "topology": "TP8, DP4 with DP attention, EP8, MTP, DSA TileLang prefill/decode",
    "repeatedTokenId": 154822,
    "badComparisonEventsBefore": 438,
    "badTilelangRowsBefore": 4914,
    "tensorAnomaliesAfterReporterPatch": 0,
    "strongOutputFailuresAfterReporterPatch": 0,
    "deterministicLocIncludesZeroAndTwo": true,
    "poisonRowUsesNan": true,
    "slotZeroSentinelSeven": true,
    "positiveSlotStillWritten": true,
    "standardAndFusedPathsSeparated": true,
    "issueExplicitlyNotFullyResolved": true,
    "aiterFollowUpLinked": true
  },
  "releaseAudit": {
    "tag": "v0.5.18",
    "resolvedSha": "71de97b264b04dcd514cf904003028aefe9775c8",
    "committedAt": "2026-08-20T21:29:24Z",
    "predatesWriterFixMerge": true,
    "releaseHasReservedSkipIndex": false,
    "issueRuntimeSha": "db570fe619ae15d8bd4ee04f146bc8234c52a20d",
    "issueRuntimeCommittedAt": "2026-08-22T13:36:46Z"
  },
  "sglangWriterFix": {
    "number": 36003,
    "title": "[Kernel] Skip reserved writes in MLA KV cache",
    "url": "https://github.com/sgl-project/sglang/pull/36003",
    "state": "closed",
    "merged": true,
    "mergedAt": "2026-08-26T05:42:24Z",
    "mergeCommitSha": "04c1036bb395cce7d9b5eab1a814163ab1dcefed",
    "mergeCommitResolvedSha": "04c1036bb395cce7d9b5eab1a814163ab1dcefed",
    "headSha": "01950f8813aecb012fb5a0a1e386bcaabfa89b24",
    "baseSha": "4b2c182d3f29e9bd35ff1271cf7bb9983c4c2c2c",
    "changedFiles": [
      "python/sglang/kernels/jit/csrc/elementwise/set_mla_kv_buffer.cuh",
      "python/sglang/kernels/ops/kvcache/mla_buffer.py",
      "python/sglang/kernels/ops/kvcache/set_mla_kv_buffer.py",
      "test/registered/kernels/benchmark/kvcache/bench_set_mla_kv_buffer.py",
      "test/registered/kernels/ops/kvcache/test_set_mla_kv_buffer.py",
      "test/registered/kernels/ops/test_kimi_k3_prerequisite_ops.py"
    ],
    "approvalCount": 1,
    "defaultReservedIndexZero": true,
    "tritonGuardsPresent": true,
    "cudaGuardPresent": true,
    "explicitMinusOneOptOut": true,
    "int32AndInt64Tests": true,
    "bf16TestPresent": true,
    "fp8TestPresent": true,
    "scaleTestPresent": true,
    "sentinelAssertions": true,
    "localGpuTestsRunByAuthor": false,
    "coversAiterFusedWriter": false
  },
  "genericStoreBoundary": {
    "number": 32477,
    "merged": true,
    "mergedAt": "2026-07-29T01:58:18Z",
    "mergeCommitSha": "ee678910f7000aa43886f218de0e159bf418f1b5",
    "reportedPassingTests": true,
    "defaultReservedIndexZero": true,
    "explicitMinusOneOptOut": true,
    "mlaWritersCovered": false
  },
  "aiterProposal": {
    "number": 5010,
    "title": "[Triton/Gluon] Support caller-defined padding cache slot in fused MLA writer",
    "url": "https://github.com/ROCm/aiter/pull/5010",
    "state": "open",
    "merged": false,
    "mergedAt": null,
    "headSha": "5448910ddc44a1539b001bf648a8ac75b8f85e98",
    "baseSha": "b12a1904fd74588e127b2deb785193e59f0ad52d",
    "changedFiles": [
      "aiter/ops/triton/_gluon_kernels/gfx1250/fusions/fused_kv_cache.py",
      "aiter/ops/triton/_triton_kernels/fusions/fused_kv_cache.py",
      "aiter/ops/triton/fusions/fused_kv_cache.py",
      "op_tests/triton_tests/fusions/test_fused_kv_cache.py"
    ],
    "approvalCount": 0,
    "defaultMinusOnePreservesBehavior": true,
    "wrapperPassesCompileTimeArgument": true,
    "guardsBothWriteBranches": true,
    "testsDefaultAndZero": true,
    "positiveSlotAssertionPresent": true,
    "sglangCallsFusedWriter": true,
    "downstreamOptInCaptured": false,
    "gpuExecutionReportedByAuthor": false
  },
  "slotContract": {
    "memoryPoolNamesSlotZeroForPaddedTokens": true,
    "memoryPoolNamesReservedPaddingSlot": true,
    "dimensions": {
      "nope": 512,
      "rope": 64
    },
    "combinedWidth": 576,
    "reservedSlot": 0,
    "validControlSlot": 2,
    "sentinelValue": 7,
    "validNopeValue": 17,
    "validRopeValue": 19
  },
  "sentinelModel": {
    "statement": "This deterministic JavaScript array model verifies the write predicate and 576-value slot contract only; it does not execute SGLang, AITER, Triton, ROCm, CUDA, or GLM-5.2.",
    "unguarded": {
      "guardReservedSlot": false,
      "reservedSlotFiniteValues": 0,
      "reservedSlotNanValues": 576,
      "reservedSlotUnchangedValues": 0,
      "validSlotNopeValues": 512,
      "validSlotRopeValues": 64
    },
    "guarded": {
      "guardReservedSlot": true,
      "reservedSlotFiniteValues": 576,
      "reservedSlotNanValues": 0,
      "reservedSlotUnchangedValues": 576,
      "validSlotNopeValues": 512,
      "validSlotRopeValues": 64
    },
    "guardPreservesAllReservedValues": true,
    "guardStillWritesAllPositiveControlValues": true,
    "unguardedPoisonsAllReservedValues": true
  },
  "deploymentGate": {
    "algorithm": [
      "Pin the exact SGLang commit, AITER commit, container digest, model revision, topology, KV dtype, and selected MLA writer before diagnosis.",
      "Run the deterministic sentinel probe with a NaN padding row mapped to slot 0 and a finite control row mapped to slot 2.",
      "Reject a standard MLA writer if the reserved_skip_index guard is absent; a version label alone is insufficient.",
      "For the fused AITER GLM path, require a merged AITER pad_slot_id contract and a pinned downstream call that explicitly passes pad_slot_id=0.",
      "Then pass finite-tensor, repeated-token, concurrency, reasoning, tool-call, long-context, restart, and rollback checks before bounded traffic."
    ],
    "fixtures": [
      {
        "name": "sglang-v0.5.18-standard-mla-writer",
        "standardWriterFixPresent": false,
        "usesFusedAiterWriter": false,
        "aiterFixMerged": false,
        "downstreamOptIn": false,
        "deterministicProbePassed": false,
        "fullRuntimeGatePassed": false,
        "decision": "reject",
        "reason": "standard-mla-writer-guard-absent"
      },
      {
        "name": "sglang-merge-04c1036-standard-writer-source-only",
        "standardWriterFixPresent": true,
        "usesFusedAiterWriter": false,
        "aiterFixMerged": false,
        "downstreamOptIn": false,
        "deterministicProbePassed": true,
        "fullRuntimeGatePassed": false,
        "decision": "hold",
        "reason": "source-fix-present-but-runtime-evidence-incomplete"
      },
      {
        "name": "glm-rocm-fused-aiter-with-sglang-fix-only",
        "standardWriterFixPresent": true,
        "usesFusedAiterWriter": true,
        "aiterFixMerged": false,
        "downstreamOptIn": false,
        "deterministicProbePassed": false,
        "fullRuntimeGatePassed": false,
        "decision": "reject",
        "reason": "fused-aiter-path-lacks-merged-end-to-end-opt-in"
      },
      {
        "name": "open-aiter-pr-without-downstream-opt-in",
        "standardWriterFixPresent": true,
        "usesFusedAiterWriter": true,
        "aiterFixMerged": false,
        "downstreamOptIn": false,
        "deterministicProbePassed": true,
        "fullRuntimeGatePassed": false,
        "decision": "reject",
        "reason": "fused-aiter-path-lacks-merged-end-to-end-opt-in"
      },
      {
        "name": "future-merged-aiter-plus-sglang-opt-in-and-runtime-gate",
        "standardWriterFixPresent": true,
        "usesFusedAiterWriter": true,
        "aiterFixMerged": true,
        "downstreamOptIn": true,
        "deterministicProbePassed": true,
        "fullRuntimeGatePassed": true,
        "decision": "eligible-for-bounded-canary",
        "reason": "pinned-guard-and-runtime-gates-pass"
      }
    ]
  },
  "decision": {
    "currentVerdict": "SGLang PR #36003 is merged for SGLang-owned MLA writers but postdates v0.5.18; it does not close the fused AITER GLM ROCm path. AITER PR #5010 remains open and the captured sources contain no downstream pad_slot_id=0 opt-in.",
    "safeAction": "Treat repeated tokens as a symptom, identify the exact writer, hold affected fused deployments, and promote only a pinned end-to-end guard that passes the deterministic probe and full runtime gates.",
    "unsafeAction": "Do not infer safety from SGLang main alone, do not apply an open PR directly to production, and do not attribute every repeated token to slot-zero poisoning without tensor and path evidence.",
    "evidenceBoundary": "Static source and a JavaScript memory model can prove the reserved-slot contract and current merge boundary, but not reproduce a GPU kernel, attribute every repeated token to this bug, or prove full-model output correctness."
  },
  "overlapAudit": {
    "prepublicationSitemapUrls": 98,
    "registryPages": 91,
    "distinctIntent": true,
    "primaryIntent": "diagnose GLM-5.2 repeated-token or NaN failures in SGLang by auditing reserved MLA KV slot 0 and the split standard-versus-fused writer fix boundary",
    "readerJob": "identify the affected writer, reject ambiguous version-only claims, run a deterministic sentinel gate, and promote only a pinned path whose exact guard and full runtime checks pass",
    "nearby": [
      [
        "https://glm52.ai/guides/glm-5-2-amd-rocm-sglang/",
        "Owns AMD hardware, checkpoint, image, and deployment-cell selection; it does not own this incident-specific slot-zero corruption gate."
      ],
      [
        "https://glm52.ai/guides/glm-5-2-hicache-kv-offload/",
        "Owns L2 and L3 KV offload economics and correctness, not physical MLA padding-slot writes."
      ],
      [
        "https://glm52.ai/guides/glm-5-2-mtp-speculative-decoding/",
        "Owns speculative-depth and acceptance tuning, not NaN poisoning of the target KV cache."
      ],
      [
        "https://glm52.ai/guides/glm-5-2-sglang-dsa-indexer-fusion/",
        "Owns indexer fusion and stream-order gates, not reserved cache-slot preservation."
      ]
    ]
  },
  "searchSupply": {
    "query": "\"GLM-5.2\" SGLang repeated tokens KV slot 0",
    "serpApiRequests": 1,
    "requestId": "serpapi-ac88bf9269a941eaa1b851793af14310",
    "knownMonthlyUsageAfter": 842,
    "result": "http-error",
    "evaluation": "failed",
    "retryCount": 0,
    "decisionImpact": "No Google evidence was returned. Topic selection and wording remain based on the committed registry, live sitemap, and pinned first-party upstream sources."
  },
  "aiHot": [
    {
      "itemId": "cmtfkvjwn0by7rou8vil9ysf1",
      "permalink": "https://aihot.virxact.com/items/cmtfkvjwn0by7rou8vil9ysf1",
      "classification": "weak-glm-link",
      "note": "The Sony and Warner complaint concerns Anthropic training data and creates no direct GLM-5.2 runtime, inference, or application decision."
    },
    {
      "itemId": "cmtf5cfxj01raro07gk66imed",
      "permalink": "https://aihot.virxact.com/items/cmtf5cfxj01raro07gk66imed",
      "classification": "weak-glm-link",
      "note": "The Uber agent-cost summary does not identify GLM-5.2 or expose a reproducible GLM-5.2 serving contract."
    }
  ],
  "discoveryChecks": {
    "zaiGlm52Release": "fetched and hashed with HTTP 200",
    "zhipuResearchIndex": "fetched and hashed after redirect with non-empty bytes"
  },
  "protectedBaseline": {
    "homepageSha256": "5d7f8327af1af910c0484cdd0b7fde8695a107339c207cc585452210f7fd05a8",
    "backlinkPageSha256": "add4d1695203254ad3ab7e420c28bd1545f18c55b86f06d17102089a2a047569",
    "homepageTitle": "GLM-5.2 Developer Guide: API, Pricing & Tools | GLM52.ai",
    "backlinkTitle": "GLM-5.2 on AMD ROCm with SGLang: A Verified Deployment Matrix | GLM52.ai"
  },
  "runtimeBoundary": {
    "modelCalls": 0,
    "modelWeightBytesDownloaded": 0,
    "sglangImports": 0,
    "aiterImports": 0,
    "cpuInferenceRuns": 0,
    "gpuRuns": 0,
    "servingProcesses": 0,
    "containers": 0,
    "thirdPartyPatchesApplied": 0,
    "note": "Static source and a JavaScript memory model can prove the reserved-slot contract and current merge boundary, but not reproduce a GPU kernel, attribute every repeated token to this bug, or prove full-model output correctness."
  },
  "invariants": {
    "twentySevenSourceReceipts": true,
    "allSourcesReturned200": true,
    "allSourceHashesValid": true,
    "immutablePinsMatch": true,
    "glmDsaContractPinned": true,
    "issueOpenAndReproducibleBoundaryCaptured": true,
    "releasePredatesMergedWriterFix": true,
    "standardWriterFixMergedAndScoped": true,
    "genericFixIsNotMlaClosure": true,
    "aiterProposalOpenAndNeedsDownstreamOptIn": true,
    "deterministicSentinelModelPasses": true,
    "fiveDeploymentFixturesFailClosed": true,
    "distinctIntent": true,
    "oneFailedSerpRequestNoRetry": true,
    "twoWeakAiHotItemsClassified": true,
    "zeroRuntimeExecution": true
  }
}
