{
  "schemaVersion": 1,
  "operationId": "20260824104145-f08e58deb8",
  "checkedAt": "2026-08-24T11:09:48Z",
  "status": "pinned-source-zero-runtime-sglang-dsa-indexer-fusion-audit",
  "revisions": {
    "sglangBeforeFeatureTag": "v0.5.14",
    "sglangBeforeFeatureCommit": "49e384ce9d304648e9959666ecb8ce8cd98d0deb",
    "sglangFirstFeatureTag": "v0.5.15",
    "sglangFirstFeatureCommit": "f63458b5beaceabbd9d749b9fc956370e1b649e6",
    "sglangAuditTag": "v0.5.18",
    "sglangAuditCommit": "71de97b264b04dcd514cf904003028aefe9775c8",
    "featureMergeCommit": "073de150530ab27e0646492ec00016c727f0d7f7",
    "streamFixMergeCommit": "6ce02b95ad9daeb476d4c1c84dde3c4b8ce735ff",
    "ropeGateFixMergeCommit": "e552f6ed75b41be838444d96319231888b4446bb",
    "glm52Fp8Revision": "ba978f7d347eaf65d22f1a86833408afdb953541"
  },
  "sourceReceipts": [
    {
      "label": "sglang-release-v0514",
      "url": "https://api.github.com/repos/sgl-project/sglang/releases/tags/v0.5.14",
      "bytes": 24702,
      "sha256": "e4593fd36878e5d25429fa2fc7875b720f4f0e928d542c9a252df861735b8e3c"
    },
    {
      "label": "sglang-release-v0515",
      "url": "https://api.github.com/repos/sgl-project/sglang/releases/tags/v0.5.15",
      "bytes": 30414,
      "sha256": "40fb0fae0d999e72be149b81ed87b5cfd83f006d759341e3314a094ac51e7a9f"
    },
    {
      "label": "sglang-release-v0518",
      "url": "https://api.github.com/repos/sgl-project/sglang/releases/tags/v0.5.18",
      "bytes": 56377,
      "sha256": "e2ad0c66ae44af55183ebd28ee2756822699df5ef800c09ae9fa31e99cdd7c6c"
    },
    {
      "label": "sglang-pr-27705",
      "url": "https://api.github.com/repos/sgl-project/sglang/pulls/27705",
      "bytes": 42883,
      "sha256": "eb75ae69f4c5b503ceb0e1635f9c1c724318431bc98a2a2953d25c2874936751"
    },
    {
      "label": "sglang-pr-29564",
      "url": "https://api.github.com/repos/sgl-project/sglang/pulls/29564",
      "bytes": 31833,
      "sha256": "652c13f0faa08808fe40fceef9260c4c0974f67d455700ed0a4b2302ea7c5e1f"
    },
    {
      "label": "sglang-pr-30018",
      "url": "https://api.github.com/repos/sgl-project/sglang/pulls/30018",
      "bytes": 20230,
      "sha256": "f87486ead1da7b8d77863cf48e4931d9d7a8f3383e462edfe60a2a47eee63bbc"
    },
    {
      "label": "sglang-pr-30025",
      "url": "https://api.github.com/repos/sgl-project/sglang/pulls/30025",
      "bytes": 27258,
      "sha256": "f0d573e4da75847ba29c8d191f468f23730542a22e7a36d9d4cd5d122e51ce28"
    },
    {
      "label": "sglang-pr-30088",
      "url": "https://api.github.com/repos/sgl-project/sglang/pulls/30088",
      "bytes": 19749,
      "sha256": "9c391db9d51a388e29381dbce90f74f0cf41e145c1eded314fbb0c91f811eec6"
    },
    {
      "label": "sglang-pr-30111",
      "url": "https://api.github.com/repos/sgl-project/sglang/pulls/30111",
      "bytes": 24415,
      "sha256": "3c847e1504dc2ab01971d8bff9006dfbb5d0a35040d6e5d73376842f477f7456"
    },
    {
      "label": "sglang-pr-30342",
      "url": "https://api.github.com/repos/sgl-project/sglang/pulls/30342",
      "bytes": 38709,
      "sha256": "6d39ccacdd0e5b79ac2c0ee21aa579250de04dee5e91d70cf0a04d59f8057722"
    },
    {
      "label": "vllm-pr-38928",
      "url": "https://api.github.com/repos/vllm-project/vllm/pulls/38928",
      "bytes": 24311,
      "sha256": "62702d7769af3bd5071b56b2cf080b8772bcf73f5650dfbc5da53dc85c1caa23"
    },
    {
      "label": "sglang-environ-v0518",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/71de97b264b04dcd514cf904003028aefe9775c8/python/sglang/srt/environ.py",
      "bytes": 86428,
      "sha256": "e64b24813f8dc9af30644ce348642adbfb4fe6365a7d6e0044347439e5ecd1ed"
    },
    {
      "label": "sglang-dsa-indexer-v0518",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/71de97b264b04dcd514cf904003028aefe9775c8/python/sglang/srt/layers/attention/dsa/dsa_indexer.py",
      "bytes": 75170,
      "sha256": "3404c9b83452d73f156b742ae32e4085a17d0b166d4d5e45afbd31cfa499b686"
    },
    {
      "label": "sglang-lora-manager-v0518",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/71de97b264b04dcd514cf904003028aefe9775c8/python/sglang/srt/lora/lora_manager.py",
      "bytes": 46242,
      "sha256": "3114bbbc37636c52b6ceb8fcfb370dfd7335891148ef8e62b97850f048efd400"
    },
    {
      "label": "sglang-weight-loader-v0518",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/71de97b264b04dcd514cf904003028aefe9775c8/python/sglang/srt/models/deepseek_common/deepseek_weight_loader.py",
      "bytes": 36096,
      "sha256": "9bb405fb6f94330394703b265cfb0cc68223fa24e063ec0314a9d9f5ba419950"
    },
    {
      "label": "sglang-fusion-test-v0518",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/71de97b264b04dcd514cf904003028aefe9775c8/test/registered/kernels/ops/attention/test_dsv32_indexer_fusion.py",
      "bytes": 10504,
      "sha256": "010813451d16c926836ab1012781f04745b43b023fae2d2e9ae968d8e4227420"
    },
    {
      "label": "sglang-glm52-cookbook-v0518",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/71de97b264b04dcd514cf904003028aefe9775c8/docs/cookbook/autoregressive/GLM/GLM-5.2.mdx",
      "bytes": 20574,
      "sha256": "ed999a20d87297acaf62a4f5c14fd0e3fec054077e71594edb630f54fcc92b11"
    },
    {
      "label": "glm52-fp8-config",
      "url": "https://huggingface.co/zai-org/GLM-5.2-FP8/resolve/ba978f7d347eaf65d22f1a86833408afdb953541/config.json",
      "bytes": 29464,
      "sha256": "22e49334abf8562fecf70ca3292ba3f5b33f5602fb2bf10b52dd64a66cfe65ff"
    },
    {
      "label": "glm52-fp8-card",
      "url": "https://huggingface.co/zai-org/GLM-5.2-FP8/resolve/ba978f7d347eaf65d22f1a86833408afdb953541/README.md",
      "bytes": 10909,
      "sha256": "de23c1b7cab43a99f0fedf4edf10de0d165882a2ecb2e2bad6c5796fcabf2e46"
    },
    {
      "label": "zai-glm52-release",
      "url": "https://z.ai/blog/glm-5.2",
      "bytes": 598,
      "sha256": "63e94b9e11a2db64243ff18418011577f2a98ca7851ce9dcb0e97cfe34cc245a"
    },
    {
      "label": "zhipu-research",
      "url": "https://www.zhipuai.cn/zh/research",
      "bytes": 1236461,
      "sha256": "188772a5cb6b65eb65f02024d89d153b692ef3d3cb2c98ddd233352c0dd6ac80"
    }
  ],
  "model": {
    "artifact": "zai-org/GLM-5.2-FP8",
    "architecture": "GlmMoeDsaForCausalLM",
    "modelType": "glm_moe_dsa",
    "decoderLayers": 78,
    "hiddenSize": 6144,
    "indexHeadDim": 128,
    "indexHeads": 32,
    "indexTopK": 2048,
    "interleavedRope": true,
    "checkpointIndexKeyProjection": "block-fp8 plus scale"
  },
  "releaseBoundary": [
    {
      "tag": "v0.5.14",
      "commit": "49e384ce9d304648e9959666ecb8ce8cd98d0deb",
      "publishedAt": "2026-06-26T22:57:30Z",
      "fusionPresent": false,
      "disableFlagDefault": null,
      "decision": "pre-feature control"
    },
    {
      "tag": "v0.5.15",
      "commit": "f63458b5beaceabbd9d749b9fc956370e1b649e6",
      "publishedAt": "2026-07-10T22:58:33Z",
      "fusionPresent": true,
      "disableFlagDefault": false,
      "decision": "first stable release containing the fusion and its follow-up stream/RoPE gates"
    },
    {
      "tag": "v0.5.18",
      "commit": "71de97b264b04dcd514cf904003028aefe9775c8",
      "publishedAt": "2026-08-22T00:09:15Z",
      "fusionPresent": true,
      "disableFlagDefault": false,
      "decision": "current pinned audit target; eligible GLM-5.2 CUDA path is on by default"
    }
  ],
  "activationMatrix": [
    {
      "platform": "CUDA",
      "ropeStyle": "GLM-5.2 interleaved",
      "disableEnvironmentValue": 0,
      "indexerTargetedLora": false,
      "result": "fusion-enabled-by-default"
    },
    {
      "platform": "CUDA",
      "ropeStyle": "GLM-5.2 interleaved",
      "disableEnvironmentValue": 1,
      "indexerTargetedLora": false,
      "result": "split-path-control"
    },
    {
      "platform": "ROCm or NPU",
      "ropeStyle": "any",
      "disableEnvironmentValue": 0,
      "indexerTargetedLora": false,
      "result": "fusion-ineligible-in-pinned-cuda-implementation"
    },
    {
      "platform": "CUDA",
      "ropeStyle": "NeoX",
      "disableEnvironmentValue": 0,
      "indexerTargetedLora": false,
      "result": "fusion-gated-off-in-v0.5.18"
    },
    {
      "platform": "CUDA",
      "ropeStyle": "GLM-5.2 interleaved",
      "disableEnvironmentValue": 0,
      "indexerTargetedLora": true,
      "result": "source-audit-blocked-before-intended-friendly-error"
    }
  ],
  "kernelReduction": {
    "source": "SGLang v0.5.15 release notes",
    "beforeKernels": 12,
    "afterKernels": 4,
    "releaseSummarySingleBatchDecodeSpeedup": "about 8%",
    "boundary": "Release-note summary; not reproduced on GLM52.ai hardware.",
    "calculatedKernelCountReductionPercent": 66.666667
  },
  "speedRows": [
    {
      "source": "SGLang pull request 27705",
      "hardware": "B300",
      "checkpoint": "GLM-5.2 FP8 command path named in the pull request",
      "batchSize": 1,
      "context": "zero context decode",
      "fusionOffTokensPerSecond": 97.72,
      "fusionOnTokensPerSecond": 107.77,
      "boundary": "Pull-request author measurement; no local GPU reproduction.",
      "calculatedThroughputChangePercent": 10.284486
    },
    {
      "source": "SGLang pull request 27705",
      "hardware": "B300",
      "checkpoint": "GLM-5.2 FP8 command path named in the pull request",
      "batchSize": 128,
      "context": "decode throughput",
      "fusionOffTokensPerSecond": 3212.05,
      "fusionOnTokensPerSecond": 3418.63,
      "boundary": "Pull-request author measurement; no local GPU reproduction.",
      "calculatedThroughputChangePercent": 6.431407
    }
  ],
  "cudaGraphHistory": {
    "disablePr": 30018,
    "streamFixPr": 30025,
    "checkpoint": "GLM-5.2-NVFP4",
    "hardwareTopology": "TP4 with EAGLE speculative decoding",
    "streamsBeforeReorder": 22,
    "streamsAfterReorder": 2,
    "headPerStepVerifyMs": 11.817,
    "fusionOffPerStepVerifyMs": 12.167,
    "fixedPerStepVerifyMs": 11.833,
    "boundary": "Pull-request author profile and ten-run benchmark; no local profiler reproduction.",
    "calculatedCapturedStreamReductionPercent": 90.909091,
    "calculatedFixedVsFusionOffVerifyCostChangePercent": -2.74513
  },
  "staticWeightMemory": {
    "scope": "GLM-5.2-FP8 indexer projection tensors only; excludes allocator, graph, workspace, KV cache, and all other model state",
    "fusedBytes": 153354240,
    "fusedMiB": 146.25,
    "unfusedBytes": 92012544,
    "unfusedMiB": 87.75,
    "deltaBytes": 61341696,
    "deltaMiB": 58.5
  },
  "reportedMemoryRows": [
    {
      "source": "SGLang pull request 29564",
      "prState": "closed-unmerged",
      "hardware": "B300",
      "tensorParallel": 4,
      "mtpEnabled": true,
      "fusionOffTargetMemoryGb": 108,
      "fusionOnTargetMemoryGb": 127,
      "fusionOffKvBudgetGb": 98,
      "fusionOnKvBudgetGb": 78,
      "fusionOffKvCapacityTokens": 1900000,
      "fusionOnKvCapacityTokens": 1520000,
      "calculatedTargetMemoryDeltaGb": 19,
      "calculatedKvBudgetDeltaGb": -20,
      "calculatedKvCapacityChangePercent": -20
    },
    {
      "source": "SGLang pull request 29564",
      "prState": "closed-unmerged",
      "hardware": "GB300",
      "tensorParallel": 4,
      "mtpEnabled": true,
      "fusionOffTargetMemoryGb": 108,
      "fusionOnTargetMemoryGb": 127,
      "fusionOffKvBudgetGb": 105,
      "fusionOnKvBudgetGb": 85,
      "fusionOffKvCapacityTokens": 2040000,
      "fusionOnKvCapacityTokens": 1660000,
      "calculatedTargetMemoryDeltaGb": 19,
      "calculatedKvBudgetDeltaGb": -20,
      "calculatedKvCapacityChangePercent": -18.627451
    },
    {
      "source": "SGLang pull request 29564",
      "prState": "closed-unmerged",
      "hardware": "B300",
      "tensorParallel": 2,
      "mtpEnabled": true,
      "fusionOffTargetMemoryGb": 211,
      "fusionOnTargetMemoryGb": 231,
      "fusionOffKvBudgetGb": 30,
      "fusionOnKvBudgetGb": 10,
      "fusionOffKvCapacityTokens": 582000,
      "fusionOnKvCapacityTokens": 199000,
      "calculatedTargetMemoryDeltaGb": 20,
      "calculatedKvBudgetDeltaGb": -20,
      "calculatedKvCapacityChangePercent": -65.80756
    },
    {
      "source": "SGLang pull request 29564",
      "prState": "closed-unmerged",
      "hardware": "GB300",
      "tensorParallel": 2,
      "mtpEnabled": true,
      "fusionOffTargetMemoryGb": 211,
      "fusionOnTargetMemoryGb": 231,
      "fusionOffKvBudgetGb": 38,
      "fusionOnKvBudgetGb": 18,
      "fusionOffKvCapacityTokens": 736000,
      "fusionOnKvCapacityTokens": 353000,
      "calculatedTargetMemoryDeltaGb": 20,
      "calculatedKvBudgetDeltaGb": -20,
      "calculatedKvCapacityChangePercent": -52.038043
    }
  ],
  "memoryReconciliation": {
    "largestReportedTargetMemoryDeltaGb": 20,
    "staticWeightDeltaMiB": 58.5,
    "staticWeightDeltaAloneExplainsReportedRuntimeDelta": false,
    "decision": "Measure total HBM, graph pools, workspaces, and KV capacity live; do not attribute an upstream runtime delta to the small fused indexer tensor alone."
  },
  "accuracyHistory": {
    "deepSeekNeoXRegression": {
      "source": "merged pull requests 30088 and 30111",
      "result": "The regression was attributed to applying interleaved-RoPE fused kernels to a NeoX-style model; v0.5.18 gates that rope style out of fusion.",
      "transferToGlm52": false
    },
    "glm52Control": {
      "source": "closed-unmerged pull request 30342",
      "checkpoint": "nvidia/GLM-5.2-NVFP4",
      "topology": "TP4 EP4",
      "fusionOnCurrentMainMean": 0.943,
      "fusionOffMean": 0.945,
      "authorConclusion": "zero material GSM8K movement for the interleaved GLM path",
      "boundary": "Reporter result in an unmerged pull request; not an in-house accuracy run or a universal parity guarantee."
    }
  },
  "loraSourceAudit": {
    "tag": "v0.5.18",
    "managerFile": "python/sglang/srt/lora/lora_manager.py",
    "indexerFile": "python/sglang/srt/layers/attention/dsa/dsa_indexer.py",
    "symbol": "_use_dsa_indexer_fusion",
    "managerImportsSymbol": true,
    "indexerDefinesSymbol": false,
    "trigger": "At least one LoRA target module intersects DSA_INDEXER_LORA_NAMES.",
    "intendedMessage": "Set SGLANG_DISABLE_DSA_INDEXER_FUSION=1 to disable fusion and use indexer LoRA.",
    "sourceAuditResult": "The pinned tag can raise ImportError before reaching the intended ValueError; the environment flag alone does not repair the missing import symbol.",
    "runtimeReproduced": false
  },
  "registeredTests": [
    {
      "file": "test/registered/kernels/ops/attention/test_dsv32_indexer_fusion.py",
      "registration": "base-b / 1-gpu-large",
      "estimatedSeconds": 45,
      "scope": "CUDA fused K normalization/RoPE/store, Q RoPE/FP8/head-gate fold, strided inputs, and grown RoPE-cache reference checks"
    }
  ],
  "runtimeBoundary": {
    "localGpuRuns": 0,
    "modelCalls": 0,
    "upstreamTestsExecutedLocally": 0,
    "sourceAuditIsRuntimeReproduction": false,
    "prMeasurementsAreGlm52aiBenchmarks": false,
    "generatedIllustrationIsProfilerTrace": false
  },
  "aiHot": {
    "fingerprint": "f1-a48ccf69ceb790e1",
    "candidateCount": 0,
    "route": "mandatory proactive research",
    "discoverySources": [
      "https://z.ai/blog/glm-5.2",
      "http://zhipuai.cn/zh/research",
      "https://github.com/sgl-project/sglang/releases/tag/v0.5.15",
      "https://github.com/sgl-project/sglang/pull/27705"
    ]
  },
  "serp": {
    "paidRequestIssued": false,
    "query": "GLM-5.2 SGLang DSA indexer fusion",
    "reason": "No paid request was needed: official release notes, pinned source, repository search, public search evidence, and the local registry established both the feature boundary and the supply gap.",
    "decisionImpact": "The page was scoped as a deployment trade-off audit rather than a generic SGLang setup guide."
  },
  "overlapAudit": {
    "prepublicationSitemapCanonicals": 87,
    "registeredScopedPages": 80,
    "exactIndexerFusionMentionsInExistingContent": 0,
    "nearestPages": [
      {
        "url": "/guides/glm-5-2-dsa-cache-layer-split/",
        "readerJob": "shard DSA cache ownership for PD prefill",
        "difference": "cache ownership and transfer are not indexer prologue kernel fusion"
      },
      {
        "url": "/guides/glm-5-2-indexshare-pipeline-splits/",
        "readerJob": "place pipeline boundaries around IndexShare",
        "difference": "layer placement is not Q/K kernel consolidation"
      },
      {
        "url": "/guides/glm-5-2-nvfp4-sglang/",
        "readerJob": "serve the NVIDIA NVFP4 checkpoint",
        "difference": "checkpoint deployment does not decide this default-on FP8/NVFP4 runtime path"
      },
      {
        "url": "/guides/glm-5-2-mtp-speculative-decoding/",
        "readerJob": "configure and validate MTP speculative decoding",
        "difference": "EAGLE is one graph-history cell here, not the primary decoding-feature setup intent"
      }
    ],
    "distinctIntent": true
  },
  "decision": {
    "classification": "cuda-default-with-a-b-canary-and-one-env-rollback",
    "eligibleProfile": "Pinned SGLang v0.5.18, GLM-5.2 interleaved RoPE, CUDA, ordinary checkpoint loading, and no indexer-targeted LoRA.",
    "promoteWhen": "Output parity passes, low and high concurrency throughput improves, total HBM and KV capacity stay within the service budget, CUDA graph stream count stays bounded, and rollback is rehearsed.",
    "rollback": "Restart the identical deployment with SGLANG_DISABLE_DSA_INDEXER_FUSION=1; environment values are read when the indexer is constructed."
  },
  "invariants": {
    "twentyOneSourceReceipts": true,
    "allSourceHashesValid": true,
    "glm52ArchitecturePinned": true,
    "releaseBeforeFeatureHasNoFusion": true,
    "firstFeatureReleaseDefaultsFusionOnForEligibleGlm": true,
    "currentReleaseDefaultsFusionOnForEligibleGlm": true,
    "releaseSummaryIsTwelveToFour": true,
    "reporterSpeedRowsPositive": true,
    "staticFp8WeightDeltaIs585MiB": true,
    "reportedRuntimeDeltaRequiresLiveReconciliation": true,
    "cudaGraphStreamCountCollapsedToTwo": true,
    "loraManagerImportsRemovedSymbol": true,
    "noLocalGpuOrModelRun": true,
    "distinctIntent": true
  }
}
