{
  "schemaVersion": 1,
  "operationId": "20260821171248-6b547ba961",
  "checkedAt": "2026-08-22T01:23:58+08:00",
  "status": "pinned-source-zero-runtime-eplb-audit",
  "scope": {
    "authenticatedApiCalls": 0,
    "liveModelCalls": 0,
    "graderCalls": 0,
    "gpuRuns": 0,
    "npuRuns": 0,
    "checkpointDownloads": 0,
    "credentialsUsed": false,
    "description": "Pinned public-source, public-manifest, and runtime-source audit with deterministic arithmetic only; no GLM, MiniMax, GPU, NPU, checkpoint shard, or serving endpoint was used."
  },
  "sourceReceipts": [
    {
      "url": "https://huggingface.co/api/models/zai-org/GLM-5.2",
      "sha256": "0d5c0e29693bf816108e1b103a68042ae167c6d10a9d2d0312606de7ab7ba542",
      "bytes": 23620,
      "role": "Observed public model revision and modification time"
    },
    {
      "url": "https://huggingface.co/zai-org/GLM-5.2/resolve/b4734de4facf877f85769a911abafc5283eab3d9/config.json",
      "sha256": "185f93ee6d12548e16a847e279dc0c3c90b1524c970b0866b42fb545747d859a",
      "bytes": 3732,
      "role": "Pinned GLM-5.2 architecture, routed-expert, dense-layer, and MTP configuration"
    },
    {
      "url": "https://huggingface.co/zai-org/GLM-5.2/resolve/b4734de4facf877f85769a911abafc5283eab3d9/model.safetensors.index.json",
      "sha256": "5fd47a926aefce0f2c917f42523e5e0f3c87e23e389e767c3681536a62f5cf5e",
      "bytes": 5408032,
      "role": "Pinned tensor-to-shard index used to count expert matrices and distinguish the MTP block"
    },
    {
      "url": "https://huggingface.co/zai-org/GLM-5.2/resolve/b4734de4facf877f85769a911abafc5283eab3d9/README.md",
      "sha256": "ed5aca8ce3dc5f8de626c87e488444343e43b1dcbdeb0e643dc72fea63ab06e8",
      "bytes": 10905,
      "role": "Pinned first-party model card and architecture boundary"
    },
    {
      "url": "https://api.github.com/repos/vllm-project/vllm/releases/latest",
      "sha256": "f9994f92dfdd57c66f58be23ecb447e9810d6d3808cf6d2d750c663f1637d24b",
      "bytes": 16784,
      "role": "Observed upstream vLLM release v0.27.1"
    },
    {
      "url": "https://github.com/vllm-project/vllm/blob/60b3d39cd36c53a698040edbf51406d3febc97a7/docs/serving/expert_parallel_deployment.md",
      "sha256": "8a9766532b5d83206de35a8993a0cd4f5fd9f303122f09294739e0814e0865f7",
      "bytes": 16860,
      "role": "Pinned vLLM expert-parallel, all-to-all, EPLB, and balancedness documentation"
    },
    {
      "url": "https://github.com/vllm-project/vllm/blob/60b3d39cd36c53a698040edbf51406d3febc97a7/vllm/config/parallel.py",
      "sha256": "e3eda322121fa41d2399a8a9ed0f2f1b1265c6b751becccf7fda456c8a516f4e",
      "bytes": 44230,
      "role": "Pinned vLLM EPLB defaults, validation, and CUDA-or-ROCm platform boundary"
    },
    {
      "url": "https://github.com/vllm-project/vllm/blob/60b3d39cd36c53a698040edbf51406d3febc97a7/vllm/distributed/eplb/eplb_state.py",
      "sha256": "7271ff3f753d01975ef90f038fbf71df4b24bbc5727bf38a76e281c5e9da2e06",
      "bytes": 52063,
      "role": "Pinned vLLM physical-to-logical expert-map initialization and rebalance state"
    },
    {
      "url": "https://github.com/vllm-project/vllm/blob/60b3d39cd36c53a698040edbf51406d3febc97a7/vllm/distributed/eplb/policy/default.py",
      "sha256": "3b54229e6240d050480390f6b29d4744fd38437d23719f78a6decf7002296a29",
      "bytes": 13765,
      "role": "Pinned vLLM policy proving physical experts must divide evenly across EP ranks"
    },
    {
      "url": "https://github.com/vllm-project/vllm/blob/60b3d39cd36c53a698040edbf51406d3febc97a7/vllm/model_executor/layers/fused_moe/expert_map_manager.py",
      "sha256": "308c00f63fcb6f44518ec3a56d8f902c7bba75928c2e3aea30fbe1f3f473973b",
      "bytes": 18886,
      "role": "Pinned vLLM expert-map manager and redundant-expert plumbing"
    },
    {
      "url": "https://github.com/vllm-project/vllm/blob/577b9623e6f8801698d411f4b04269326f5afbe2/docs/serving/data_parallel_deployment.md",
      "sha256": "bb4f301d950b3f19127be7c93dcbe2033cc75859a998586782dd1ca0ee52749c",
      "bytes": 9489,
      "role": "Pinned vLLM data-parallel process and per-rank scheduling semantics"
    },
    {
      "url": "https://api.github.com/repos/sgl-project/sglang/releases/latest",
      "sha256": "7ff5c5b75f1a42221c14f053bed152790343bf9feab5f902e7f954a1f2f476a7",
      "bytes": 66575,
      "role": "Observed SGLang release v0.5.17"
    },
    {
      "url": "https://github.com/sgl-project/sglang/blob/22dde1dd5b56d648251c115498fd4e1815a6264a/docs/cookbook/autoregressive/GLM/GLM-5.2.mdx",
      "sha256": "ed999a20d87297acaf62a4f5c14fd0e3fec054077e71594edb630f54fcc92b11",
      "bytes": 20574,
      "role": "Pinned GLM-5.2-specific SGLang cookbook and DP-Attention plus DeepEP guidance"
    },
    {
      "url": "https://github.com/sgl-project/sglang/blob/b819d2fb5bbdff3dfb969b4abd678cf54799e405/docs/docs/advanced_features/expert_parallelism.mdx",
      "sha256": "39cce8ce61ee691d9c54a11ffeec82c558f83fe96a8e12bd945469bae2d0a429",
      "bytes": 21578,
      "role": "Pinned SGLang DeepEP modes, backend constraints, and EPLB description"
    },
    {
      "url": "https://github.com/deepseek-ai/EPLB/blob/d52c72d5b2f2fb4c41afbf8eb21366820239913d/README.md",
      "sha256": "897fa18b928858d6f4241fa910377125fbbc16fba8868b8c121eb5a8c30336ef",
      "bytes": 3131,
      "role": "Pinned reference algorithm for redundant expert placement and hierarchical or global balancing"
    },
    {
      "url": "https://z.ai/blog/glm-5.2",
      "sha256": "63e94b9e11a2db64243ff18418011577f2a98ca7851ce9dcb0e97cfe34cc245a",
      "bytes": 598,
      "role": "Fixed per-round Z.AI GLM-5.2 release-page check"
    },
    {
      "url": "https://www.zhipuai.cn/zh/research",
      "sha256": "188772a5cb6b65eb65f02024d89d153b692ef3d3cb2c98ddd233352c0dd6ac80",
      "bytes": 1236461,
      "role": "Fixed per-round Zhipu research-index check"
    }
  ],
  "revisions": {
    "glm52": "b4734de4facf877f85769a911abafc5283eab3d9",
    "glm52LastModified": "2026-07-02T08:08:14.000Z",
    "vllmRelease": "v0.27.1",
    "vllmReleasePublishedAt": "2026-08-11T10:47:49Z",
    "vllmExpertParallelDocs": "60b3d39cd36c53a698040edbf51406d3febc97a7",
    "vllmDataParallelDocs": "577b9623e6f8801698d411f4b04269326f5afbe2",
    "sglangRelease": "v0.5.17",
    "sglangReleasePublishedAt": "2026-08-08T00:19:16Z",
    "sglangGlm52Cookbook": "22dde1dd5b56d648251c115498fd4e1815a6264a",
    "sglangExpertParallelDocs": "b819d2fb5bbdff3dfb969b4abd678cf54799e405",
    "deepseekEplb": "d52c72d5b2f2fb4c41afbf8eb21366820239913d"
  },
  "architecture": {
    "modelType": "glm_moe_dsa",
    "hiddenSize": 6144,
    "moeIntermediateSize": 2048,
    "numHiddenLayers": 78,
    "firstKDenseReplace": 3,
    "numNextnPredictLayers": 1,
    "numRoutedExperts": 256,
    "numExpertsPerToken": 8,
    "numSharedExperts": 1,
    "expertLinearProjections": 3,
    "checkpointTensorCount": 59585,
    "checkpointExpertTensorCount": 58368,
    "checkpointExpertBearingLayerFirst": 3,
    "checkpointExpertBearingLayerLast": 78,
    "checkpointExpertBearingLayerCount": 76,
    "baseTransformerExpertBearingLayerCount": 75,
    "checkpointExpertTensorsPerProjection": 19456,
    "checkpointTotalSizeBytes": 1506659919872
  },
  "architectureArithmetic": {
    "baseMoeLayerCount": 75,
    "checkpointMoeLayerCount": 76,
    "expertParametersPerLayer": 37748736,
    "bf16BytesPerExpertPerLayer": 75497472,
    "bf16GiBPerExpertPerLayer": 0.070313,
    "baseExpertFamilyParameters": 2831155200,
    "baseExpertFamilyBf16Bytes": 5662310400,
    "baseExpertFamilyBf16GiB": 5.273438,
    "checkpointExpertFamilyParameters": 2868903936,
    "checkpointExpertFamilyBf16Bytes": 5737807872,
    "checkpointExpertFamilyBf16GiB": 5.34375,
    "baseRoutedLibraryBf16Bytes": 1449551462400,
    "baseRoutedLibraryBf16GiB": 1350,
    "checkpointRoutedLibraryBf16Bytes": 1468878815232,
    "checkpointRoutedLibraryBf16GiB": 1368,
    "checkpointRoutedLibraryShareOfPublishedTensorBytesPercent": 97.492,
    "activeRoutedExpertFraction": 0.03125,
    "activeRoutedExpertPercent": 3.125,
    "activeBaseExpertFamilyBf16GiB": 42.1875,
    "observedExpertTensorIdentity": 58368
  },
  "topologyRows": [
    {
      "epSize": 8,
      "baseSlotsPerRank": 32,
      "baseTransformerRoutedBf16GiBPerRank": 168.75,
      "checkpointRoutedBf16GiBPerRank": 171,
      "oneExtraSlotPerRank": {
        "globalRedundantCount": 8,
        "totalPhysicalExperts": 264,
        "slotsPerRank": 33,
        "expertPayloadOverheadPercent": 3.125,
        "baseTransformerBf16GiBPerRankAdded": 5.273438,
        "checkpointBf16GiBPerRankAdded": 5.34375
      },
      "genericR32": {
        "globalRedundantCount": 32,
        "totalPhysicalExperts": 288,
        "dividesEvenly": true,
        "slotsPerRank": 36,
        "addedSlotsPerRank": 4,
        "baseTransformerBf16GiBPerRankAdded": 21.09375,
        "checkpointBf16GiBPerRankAdded": 21.375,
        "expertPayloadOverheadPercent": 12.5
      }
    },
    {
      "epSize": 16,
      "baseSlotsPerRank": 16,
      "baseTransformerRoutedBf16GiBPerRank": 84.375,
      "checkpointRoutedBf16GiBPerRank": 85.5,
      "oneExtraSlotPerRank": {
        "globalRedundantCount": 16,
        "totalPhysicalExperts": 272,
        "slotsPerRank": 17,
        "expertPayloadOverheadPercent": 6.25,
        "baseTransformerBf16GiBPerRankAdded": 5.273438,
        "checkpointBf16GiBPerRankAdded": 5.34375
      },
      "genericR32": {
        "globalRedundantCount": 32,
        "totalPhysicalExperts": 288,
        "dividesEvenly": true,
        "slotsPerRank": 18,
        "addedSlotsPerRank": 2,
        "baseTransformerBf16GiBPerRankAdded": 10.546875,
        "checkpointBf16GiBPerRankAdded": 10.6875,
        "expertPayloadOverheadPercent": 12.5
      }
    },
    {
      "epSize": 32,
      "baseSlotsPerRank": 8,
      "baseTransformerRoutedBf16GiBPerRank": 42.1875,
      "checkpointRoutedBf16GiBPerRank": 42.75,
      "oneExtraSlotPerRank": {
        "globalRedundantCount": 32,
        "totalPhysicalExperts": 288,
        "slotsPerRank": 9,
        "expertPayloadOverheadPercent": 12.5,
        "baseTransformerBf16GiBPerRankAdded": 5.273438,
        "checkpointBf16GiBPerRankAdded": 5.34375
      },
      "genericR32": {
        "globalRedundantCount": 32,
        "totalPhysicalExperts": 288,
        "dividesEvenly": true,
        "slotsPerRank": 9,
        "addedSlotsPerRank": 1,
        "baseTransformerBf16GiBPerRankAdded": 5.273438,
        "checkpointBf16GiBPerRankAdded": 5.34375,
        "expertPayloadOverheadPercent": 12.5
      }
    },
    {
      "epSize": 64,
      "baseSlotsPerRank": 4,
      "baseTransformerRoutedBf16GiBPerRank": 21.09375,
      "checkpointRoutedBf16GiBPerRank": 21.375,
      "oneExtraSlotPerRank": {
        "globalRedundantCount": 64,
        "totalPhysicalExperts": 320,
        "slotsPerRank": 5,
        "expertPayloadOverheadPercent": 25,
        "baseTransformerBf16GiBPerRankAdded": 5.273438,
        "checkpointBf16GiBPerRankAdded": 5.34375
      },
      "genericR32": {
        "globalRedundantCount": 32,
        "totalPhysicalExperts": 288,
        "dividesEvenly": false,
        "slotsPerRank": null,
        "addedSlotsPerRank": null,
        "baseTransformerBf16GiBPerRankAdded": null,
        "checkpointBf16GiBPerRankAdded": null,
        "expertPayloadOverheadPercent": 12.5
      }
    },
    {
      "epSize": 128,
      "baseSlotsPerRank": 2,
      "baseTransformerRoutedBf16GiBPerRank": 10.546875,
      "checkpointRoutedBf16GiBPerRank": 10.6875,
      "oneExtraSlotPerRank": {
        "globalRedundantCount": 128,
        "totalPhysicalExperts": 384,
        "slotsPerRank": 3,
        "expertPayloadOverheadPercent": 50,
        "baseTransformerBf16GiBPerRankAdded": 5.273438,
        "checkpointBf16GiBPerRankAdded": 5.34375
      },
      "genericR32": {
        "globalRedundantCount": 32,
        "totalPhysicalExperts": 288,
        "dividesEvenly": false,
        "slotsPerRank": null,
        "addedSlotsPerRank": null,
        "baseTransformerBf16GiBPerRankAdded": null,
        "checkpointBf16GiBPerRankAdded": null,
        "expertPayloadOverheadPercent": 12.5
      }
    },
    {
      "epSize": 256,
      "baseSlotsPerRank": 1,
      "baseTransformerRoutedBf16GiBPerRank": 5.273438,
      "checkpointRoutedBf16GiBPerRank": 5.34375,
      "oneExtraSlotPerRank": {
        "globalRedundantCount": 256,
        "totalPhysicalExperts": 512,
        "slotsPerRank": 2,
        "expertPayloadOverheadPercent": 100,
        "baseTransformerBf16GiBPerRankAdded": 5.273438,
        "checkpointBf16GiBPerRankAdded": 5.34375
      },
      "genericR32": {
        "globalRedundantCount": 32,
        "totalPhysicalExperts": 288,
        "dividesEvenly": false,
        "slotsPerRank": null,
        "addedSlotsPerRank": null,
        "baseTransformerBf16GiBPerRankAdded": null,
        "checkpointBf16GiBPerRankAdded": null,
        "expertPayloadOverheadPercent": 12.5
      }
    }
  ],
  "runtime": {
    "vllmEpNormalWorldSizeRule": "TP x DP; current validation also includes prefill-context-parallel ranks",
    "vllmEnableEpFlag": "--enable-expert-parallel",
    "vllmEnableEplbFlag": "--enable-eplb",
    "vllmEplbDefaults": {
      "windowSize": 1000,
      "stepInterval": 3000,
      "numRedundantExperts": 0,
      "logBalancedness": false,
      "useAsync": true
    },
    "vllmSupportedEplbPlatformsAtPinnedSource": [
      "CUDA",
      "ROCm"
    ],
    "vllmGenericLargeScaleRedundantExpertSuggestion": 32,
    "sglangGlmSpecificRoute": "DP-Attention plus DeepEP for balanced and high-throughput strategies",
    "sglangDispatchModes": [
      "normal",
      "low_latency",
      "auto"
    ],
    "sglangPeriodicRebalanceExampleRequests": 1000,
    "balancednessDefinition": "mean expert token load divided by maximum expert token load"
  },
  "boundaries": {
    "runtimeBenchmarksPerformed": false,
    "glmSpecificEplbGainMeasured": false,
    "genericRedundancySuggestionIsGlmSpecific": false,
    "bf16MemoryArithmeticIncludesRuntimeOverheads": false,
    "quantizedRedundancyCanBeDerivedByDividingBf16Bytes": false,
    "vllmDocsRedundantExpertLabelIsUnambiguous": false,
    "reasonForVllmLabelBoundary": "The parameter table says additional global experts per EP rank, while the pinned state and policy source represent one global physical-expert pool with logical experts plus a global redundant count.",
    "glmConfigExpertGroupCount": 1,
    "glmConfigTopkGroupCount": 1,
    "ascendCoveredByPinnedVllmCoreEplbValidation": false
  },
  "decisionRules": [
    "Start with expert parallelism and no redundant experts. Add EPLB only after the production-shaped workload shows persistent per-rank or per-expert skew.",
    "Treat num_redundant_experts as a global count in the pinned vLLM source: total physical experts equal 256 logical experts plus the redundant count, and that total must divide evenly across EP ranks.",
    "Do not copy the generic r=32 suggestion without checking divisibility and memory. It is valid for EP8, EP16, and EP32 in this matrix, but not EP64, EP128, or EP256.",
    "Budget 5.273438 GiB of BF16 routed-expert weights per rank for one extra slot across the 75 base MoE layers, or 5.34375 GiB if the runtime also replicates the checkpoint MTP expert-bearing block; measure the loaded graph to select the correct bound.",
    "Do not divide the BF16 structural estimate by two or four to claim FP8 or 4-bit process fit. Quantized layouts, scales, shared experts, attention, KV cache, communication workspaces, graphs, and allocator reserve remain runtime-specific.",
    "Measure balancedness, per-rank token load, all-to-all time, TTFT, TPOT, throughput, peak memory, quality, parser behavior, and recovery on the same request mix before promotion."
  ],
  "invariants": {
    "seventeenSourcesHashed": true,
    "noModelsOrAcceleratorsUsed": true,
    "baseMoeLayerCountIsSeventyFive": true,
    "checkpointIncludesOneExtraMtpExpertBlock": true,
    "expertTensorCountIdentityMatches": true,
    "threeProjectionCountsMatch": true,
    "activeRoutedFractionIsThreePointOneTwoFivePercent": true,
    "baseExpertFamilyIsFivePointTwoSevenGiB": true,
    "checkpointExpertFamilyIsFivePointThreeFourGiB": true,
    "everyBaseTopologyDividesLogicalExperts": true,
    "oneExtraSlotPerRankAlwaysDivides": true,
    "genericR32ValidityIsExplicit": true,
    "platformBoundaryFailsClosed": true,
    "noGenericGlmPerformanceClaim": true
  },
  "limits": [
    "This is a structural capacity and source-contract audit, not a GLM-5.2 inference benchmark. No checkpoint shard or accelerator was used.",
    "The BF16 per-slot figures describe routed expert matrices only. They exclude shared experts, dense layers, attention, embeddings, MTP auxiliaries, KV cache, graphs, dispatch buffers, allocator fragmentation, replicas, and failover reserve.",
    "The base 75-layer and checkpoint-wide 76-block figures intentionally differ. The extra checkpoint block is the configured next-token-prediction layer; whether a serving runtime includes it in an EPLB model state must be verified from the loaded graph.",
    "vLLM documentation contains ambiguous wording for num_redundant_experts. The arithmetic follows the pinned implementation, which builds one global physical-to-logical list of 256 logical plus r redundant experts.",
    "SGLang documents GLM-5.2 with DP-Attention and DeepEP, and documents EPLB generally. That combination does not by itself prove a throughput gain or a safe rebalance cadence for a particular GLM-5.2 workload.",
    "Current release observations and project main-branch source are dated receipts, not promises that future CLI flags, defaults, platform support, or policies will remain unchanged."
  ]
}
