{
  "schemaVersion": 1,
  "operationId": "20260822191948-e20bb070ea",
  "checkedAt": "2026-08-22T19:25:00Z",
  "status": "pinned-source-zero-runtime-layer-split-audit",
  "revisions": {
    "sglangTag": "v0.5.18",
    "sglangCommit": "71de97b264b04dcd514cf904003028aefe9775c8",
    "featureReleaseTag": "v0.5.16",
    "featureMergeCommit": "8e54517f027606a00054fc25c12898bc82acc65b",
    "glm52Fp8Revision": "ba978f7d347eaf65d22f1a86833408afdb953541"
  },
  "sourceReceipts": [
    {
      "label": "sglang-release-v0516",
      "url": "https://api.github.com/repos/sgl-project/sglang/releases/tags/v0.5.16",
      "bytes": 32424,
      "sha256": "c44d35acbb9e932ba4977721dbfabdb710f7506ca3d34c0b5dd64787d5677c1a"
    },
    {
      "label": "sglang-release-v0518",
      "url": "https://api.github.com/repos/sgl-project/sglang/releases/tags/v0.5.18",
      "bytes": 53374,
      "sha256": "2abaea076aa8553329dddd4201501ed94edde801dd93ca47669ef3a9de2ef8bf"
    },
    {
      "label": "sglang-server-args",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/71de97b264b04dcd514cf904003028aefe9775c8/python/sglang/srt/server_args.py",
      "bytes": 444424,
      "sha256": "5ed5badef8dca0ee3dd2b34a1c9c417e172ff1a947d9fe802f7983fb4a2be950"
    },
    {
      "label": "sglang-cp-utils",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/71de97b264b04dcd514cf904003028aefe9775c8/python/sglang/srt/layers/cp/utils.py",
      "bytes": 12012,
      "sha256": "7cf1c1d05c999fe570fe297f899a147ef20b118f052df49ec530de27d8facb25"
    },
    {
      "label": "sglang-layer-pool",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/71de97b264b04dcd514cf904003028aefe9775c8/python/sglang/srt/mem_cache/dsa_cache_layer_split.py",
      "bytes": 24029,
      "sha256": "5cf66578e453fd75fb5149a4f0e4698e27c7f81f5c66ff4955f41aa9cdc27dd7"
    },
    {
      "label": "sglang-pd-conn",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/71de97b264b04dcd514cf904003028aefe9775c8/python/sglang/srt/disaggregation/common/conn.py",
      "bytes": 79084,
      "sha256": "dc28d75e40b63163334b78e097b5eaa07ed768d596ce4c3903297cd20a3debf3"
    },
    {
      "label": "sglang-pd-prefill",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/71de97b264b04dcd514cf904003028aefe9775c8/python/sglang/srt/disaggregation/prefill.py",
      "bytes": 56841,
      "sha256": "32ed50d6faad38bc9b7a4891420a370ef4d22d34631ae7dfccdaba104f9327da"
    },
    {
      "label": "sglang-glm52-cookbook",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/71de97b264b04dcd514cf904003028aefe9775c8/docs/cookbook/autoregressive/GLM/GLM-5.2.mdx",
      "bytes": 20574,
      "sha256": "ed999a20d87297acaf62a4f5c14fd0e3fec054077e71594edb630f54fcc92b11"
    },
    {
      "label": "sglang-layer-cpu-test",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/71de97b264b04dcd514cf904003028aefe9775c8/test/registered/unit/mem_cache/test_dsa_layer_shard_utils.py",
      "bytes": 5029,
      "sha256": "270f9cb9d046b7b7bd82f171d1e1d7c31c7e06bef235e7e74e6873bec8021f88"
    },
    {
      "label": "sglang-layer-gpu-test",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/71de97b264b04dcd514cf904003028aefe9775c8/test/registered/unit/mem_cache/test_dsa_layer_split_broadcast.py",
      "bytes": 5469,
      "sha256": "3613a9db5088fd748e023f46a51c51feead319954e53efb9fa3575e411717227"
    },
    {
      "label": "sglang-layer-e2e-test",
      "url": "https://raw.githubusercontent.com/sgl-project/sglang/71de97b264b04dcd514cf904003028aefe9775c8/test/registered/models_e2e/test_dsa_glm52_pd_mtp_cp_layersplit.py",
      "bytes": 2931,
      "sha256": "a7cad9b7390ad7364bdc67f1efd6830ab70cf8db6d4cb045b4cf1db9c6921076"
    },
    {
      "label": "sglang-pr-29421",
      "url": "https://api.github.com/repos/sgl-project/sglang/pulls/29421",
      "bytes": 50012,
      "sha256": "2e8ef4c041b8ad6dd3407a312148dd4533a60838c7625ed5995ca1a94b9165b6"
    },
    {
      "label": "sglang-pr-29161",
      "url": "https://api.github.com/repos/sgl-project/sglang/pulls/29161",
      "bytes": 30295,
      "sha256": "9af97d2711a9a10b7e99b784cbc689e652d0580a6117c173b3fefe35ae935288"
    },
    {
      "label": "sglang-pr-29166",
      "url": "https://api.github.com/repos/sgl-project/sglang/pulls/29166",
      "bytes": 22415,
      "sha256": "b55346f60ca8854006ac978bc6865abcfcc34d7048555e7b29d80aac3306bbfe"
    },
    {
      "label": "glm52-fp8-config",
      "url": "https://huggingface.co/zai-org/GLM-5.2-FP8/resolve/ba978f7d347eaf65d22f1a86833408afdb953541/config.json",
      "bytes": 29464,
      "sha256": "22e49334abf8562fecf70ca3292ba3f5b33f5602fb2bf10b52dd64a66cfe65ff"
    },
    {
      "label": "glm52-fp8-model-card",
      "url": "https://huggingface.co/zai-org/GLM-5.2-FP8/resolve/ba978f7d347eaf65d22f1a86833408afdb953541/README.md",
      "bytes": 10909,
      "sha256": "de23c1b7cab43a99f0fedf4edf10de0d165882a2ecb2e2bad6c5796fcabf2e46"
    },
    {
      "label": "zai-glm52-release",
      "url": "https://z.ai/blog/glm-5.2",
      "bytes": 598,
      "sha256": "63e94b9e11a2db64243ff18418011577f2a98ca7851ce9dcb0e97cfe34cc245a"
    },
    {
      "label": "zai-layer-split",
      "url": "https://z.ai/blog/scaling-pain",
      "bytes": 603,
      "sha256": "537a640b69319af3b19e57c8e970016d51264e608f1ca96dd37e9012a596efb8"
    },
    {
      "label": "zhipu-research",
      "url": "https://www.zhipuai.cn/zh/research",
      "bytes": 1236461,
      "sha256": "188772a5cb6b65eb65f02024d89d153b692ef3d3cb2c98ddd233352c0dd6ac80"
    }
  ],
  "model": {
    "artifact": "zai-org/GLM-5.2-FP8",
    "architecture": "GlmMoeDsaForCausalLM",
    "decoderLayers": 78,
    "mtpLayers": 1,
    "maxPositionEmbeddings": 1048576,
    "indexTopK": 2048,
    "indexHeadDim": 128,
    "kvLoraRank": 512,
    "qkRopeHeadDim": 64
  },
  "publishedMeasurement": {
    "source": "SGLang v0.5.16 release notes and pull request 29421",
    "artifact": "GLM-5.2-FP8",
    "tokens": 8192,
    "cpSize": 4,
    "baselineGbPerRank": 0.77,
    "layerSplitGbPerRank": 0.2,
    "reportedReduction": "about 74%",
    "transferableBeyondNamedCell": false,
    "calculatedReductionPercent": 74.026,
    "aggregateBaselineGbAcrossFourRanks": 3.08,
    "aggregateLayerSplitGbAcrossFourRanks": 0.8,
    "boundary": "The named 8,192-token cache cell is not total server HBM, weight memory, a throughput result, or a guarantee for another CP size."
  },
  "layerPlans": [
    {
      "cpSize": 2,
      "ranges": [
        {
          "rank": 0,
          "start": 0,
          "end": 39,
          "ownedLayers": 39
        },
        {
          "rank": 1,
          "start": 39,
          "end": 78,
          "ownedLayers": 39
        }
      ],
      "ownedLayers": [
        39,
        39
      ],
      "minOwnedLayers": 39,
      "maxOwnedLayers": 39,
      "imbalanceLayers": 0,
      "effectiveLayerEquivalents": 40,
      "ownedOnlyUpperBoundReductionPercent": 50,
      "scratchIncludedUpperBoundReductionPercent": 48.7179,
      "completeCoverage": true
    },
    {
      "cpSize": 4,
      "ranges": [
        {
          "rank": 0,
          "start": 0,
          "end": 20,
          "ownedLayers": 20
        },
        {
          "rank": 1,
          "start": 20,
          "end": 40,
          "ownedLayers": 20
        },
        {
          "rank": 2,
          "start": 40,
          "end": 59,
          "ownedLayers": 19
        },
        {
          "rank": 3,
          "start": 59,
          "end": 78,
          "ownedLayers": 19
        }
      ],
      "ownedLayers": [
        20,
        20,
        19,
        19
      ],
      "minOwnedLayers": 19,
      "maxOwnedLayers": 20,
      "imbalanceLayers": 1,
      "effectiveLayerEquivalents": 21,
      "ownedOnlyUpperBoundReductionPercent": 74.359,
      "scratchIncludedUpperBoundReductionPercent": 73.0769,
      "completeCoverage": true
    },
    {
      "cpSize": 8,
      "ranges": [
        {
          "rank": 0,
          "start": 0,
          "end": 10,
          "ownedLayers": 10
        },
        {
          "rank": 1,
          "start": 10,
          "end": 20,
          "ownedLayers": 10
        },
        {
          "rank": 2,
          "start": 20,
          "end": 30,
          "ownedLayers": 10
        },
        {
          "rank": 3,
          "start": 30,
          "end": 40,
          "ownedLayers": 10
        },
        {
          "rank": 4,
          "start": 40,
          "end": 50,
          "ownedLayers": 10
        },
        {
          "rank": 5,
          "start": 50,
          "end": 60,
          "ownedLayers": 10
        },
        {
          "rank": 6,
          "start": 60,
          "end": 69,
          "ownedLayers": 9
        },
        {
          "rank": 7,
          "start": 69,
          "end": 78,
          "ownedLayers": 9
        }
      ],
      "ownedLayers": [
        10,
        10,
        10,
        10,
        10,
        10,
        9,
        9
      ],
      "minOwnedLayers": 9,
      "maxOwnedLayers": 10,
      "imbalanceLayers": 1,
      "effectiveLayerEquivalents": 11,
      "ownedOnlyUpperBoundReductionPercent": 87.1795,
      "scratchIncludedUpperBoundReductionPercent": 85.8974,
      "completeCoverage": true
    }
  ],
  "requirements": [
    {
      "id": "dsa-model",
      "required": "DeepSeek Sparse Attention model",
      "glm52Result": "pass",
      "evidence": "The pinned GLM-5.2-FP8 config declares GlmMoeDsaForCausalLM."
    },
    {
      "id": "pd-prefill-role",
      "required": "PD-disaggregated prefill worker",
      "glm52Result": "conditional",
      "evidence": "SGLang rejects the flag on unified and decode workers."
    },
    {
      "id": "prefill-cp",
      "required": "--enable-prefill-cp",
      "glm52Result": "conditional",
      "evidence": "Server validation requires prefill context parallelism."
    },
    {
      "id": "interleave",
      "required": "--cp-strategy interleave",
      "glm52Result": "conditional",
      "evidence": "Zigzag and an unset strategy fail validation."
    },
    {
      "id": "cp-size",
      "required": "attention CP size greater than one",
      "glm52Result": "conditional",
      "evidence": "The runtime returns ordinary non-sharded semantics at CP1."
    },
    {
      "id": "mooncake",
      "required": "Mooncake or Mooncake TCP PD transfer",
      "glm52Result": "conditional",
      "evidence": "The pinned server rejects Mori and NIXL for this feature."
    },
    {
      "id": "pp-one",
      "required": "pipeline parallel size one",
      "glm52Result": "conditional",
      "evidence": "The pinned server rejects pp_size greater than one."
    },
    {
      "id": "ordinary-decode-cache",
      "required": "decode worker keeps ordinary full local cache semantics",
      "glm52Result": "conditional",
      "evidence": "The decode worker receives shards from all prefill CP ranks and must not enable LayerSplit."
    },
    {
      "id": "draft-full-cache",
      "required": "MTP draft worker remains unsharded",
      "glm52Result": "pass-by-design",
      "evidence": "is_glm_dsa_cache_layer_split_enabled excludes draft workers."
    }
  ],
  "configCases": [
    {
      "case": "PD prefill, DSA, CP4, interleave, Mooncake, PP1",
      "outcome": "eligible-for-bounded-pilot"
    },
    {
      "case": "unified prefill plus decode worker",
      "outcome": "startup-reject"
    },
    {
      "case": "PD decode worker with LayerSplit flag",
      "outcome": "startup-reject"
    },
    {
      "case": "PD prefill with zigzag CP",
      "outcome": "startup-reject"
    },
    {
      "case": "PD prefill with NIXL or Mori transfer",
      "outcome": "startup-reject"
    },
    {
      "case": "PD prefill with pipeline parallel size two",
      "outcome": "startup-reject"
    },
    {
      "case": "PD prefill with CP1",
      "outcome": "starts-without-layer-sharding"
    },
    {
      "case": "non-DSA model",
      "outcome": "startup-reject"
    }
  ],
  "registeredTests": [
    {
      "scope": "CPU partition and scratch-path unit tests",
      "file": "test/registered/unit/mem_cache/test_dsa_layer_shard_utils.py",
      "registration": "base-a-test-cpu",
      "estimatedSeconds": 1
    },
    {
      "scope": "multi-GPU owner-broadcast integration test",
      "file": "test/registered/unit/mem_cache/test_dsa_layer_split_broadcast.py",
      "registration": "base-c / 4-gpu-b200",
      "estimatedSeconds": 120
    },
    {
      "scope": "GLM-5.2 PD plus MTP GSM8K end-to-end test",
      "file": "test/registered/models_e2e/test_dsa_glm52_pd_mtp_cp_layersplit.py",
      "registration": "base-c / 8-gpu-b300",
      "estimatedSeconds": 750,
      "questions": 1319,
      "accuracyFloor": 0.935,
      "shots": 20
    }
  ],
  "runtimeBoundary": {
    "localGpuRuns": 0,
    "modelCalls": 0,
    "registeredTestPresenceProvesExecution": false,
    "publishedMeasurementIsWholeServerHbm": false,
    "weightsAffected": false,
    "decodeCacheSharded": false,
    "ownerBroadcastAndPdTransferRequireLiveValidation": true
  },
  "serp": {
    "requestId": "serpapi-ca9abb9fb289458b8f4f8cc2f102157f",
    "query": "GLM-5.2 DSA cache layer split",
    "organicResults": 9,
    "officialSglangPosition": 4,
    "value": "high-value",
    "decisionImpact": "Confirmed an official configuration result but no independent GLM-5.2 deployment audit, supporting a distinct fail-closed reader job."
  },
  "overlapAudit": {
    "prepublicationSitemapCanonicals": 84,
    "registeredScopedPages": 77,
    "exactFeatureMentionsInExistingContent": 0,
    "nearestPages": [
      "/guides/glm-5-2-indexshare-pipeline-splits/",
      "/guides/glm-5-2-expert-parallelism/",
      "/guides/glm-5-2-sglang-weight-cache-daemon/",
      "/guides/glm-5-2-amd-rocm-sglang/"
    ],
    "distinctIntent": true
  },
  "decision": {
    "classification": "bounded-pd-prefill-pilot-only",
    "eligibleProfile": "Pinned SGLang v0.5.18; GLM-5.2 DSA; PD prefill role; prefill CP interleave; CP greater than one; Mooncake transfer; PP1; ordinary decode cache.",
    "promotionGate": "Require startup validation, shard-plan logs, all-rank PD registration, owner-broadcast parity, fixed-corpus answer parity, cache-memory telemetry, TTFT/throughput distributions, and rollback evidence.",
    "rejectWhen": "Unified serving, decode-side flag, zigzag CP, CP1 expected to save memory, NIXL/Mori transfer, PP greater than one, non-DSA models, or an unpinned build."
  },
  "invariants": {
    "nineteenSourceReceipts": true,
    "allSourceHashesValid": true,
    "glm52IsDsa": true,
    "decoderLayerCountIs78": true,
    "everyPlanCoversEachLayerExactlyOnce": true,
    "everyPlanImbalanceAtMostOneLayer": true,
    "cp4RangesMatchPinnedFormula": true,
    "publishedReductionRoundsToAbout74Percent": true,
    "registeredTestsAreCoverageNotExecution": true,
    "noLocalGpuOrModelRun": true,
    "distinctIntent": true
  }
}
