{
  "schemaVersion": 1,
  "operationId": "20260821041949-9899ee633f",
  "checkedAt": "2026-08-21T12:32:36+08:00",
  "status": "pinned-source-zero-npu-audit",
  "scope": {
    "authenticatedApiCalls": 0,
    "liveModelCalls": 0,
    "graderCalls": 0,
    "npuRuns": 0,
    "gpuRuns": 0,
    "checkpointDownloads": 0,
    "credentialsUsed": false,
    "description": "Pinned public-source and public-manifest audit with deterministic arithmetic only; no GLM, MiniMax, NPU, GPU, model checkpoint, or serving endpoint was used."
  },
  "sourceReceipts": [
    {
      "url": "https://github.com/vllm-project/vllm-ascend/blob/5cb98caaadeff42b5b62b996e34bb2aaa29d20fd/docs/source/tutorials/models/GLM5.2.md",
      "sha256": "df3d1414467af2453c6cdfdb124fd3727a103862ead73a8546c7a02845d848bb",
      "bytes": 72365,
      "role": "Pinned vLLM Ascend v0.23.0 GLM-5.2 deployment scenarios and tuning tables"
    },
    {
      "url": "https://github.com/vllm-project/vllm-ascend/blob/5cb98caaadeff42b5b62b996e34bb2aaa29d20fd/docs/source/user_guide/support_matrix/supported_models.md",
      "sha256": "f25bcf4993db14e0767962e0265c15fc507a0974cfb13ce79c390c2b85f3e055",
      "bytes": 22020,
      "role": "Pinned release support status, feature flags, and 200k matrix entry"
    },
    {
      "url": "https://github.com/vllm-project/vllm-ascend/blob/b25f49dd0281726017041eed92bd0edc56961a2e/docs/source/tutorials/models/GLM5.2.md",
      "sha256": "49931618646bfb4bae633345ab749702ad2444da12497c552656f761bc67b465",
      "bytes": 72724,
      "role": "Pinned current tutorial with corrected A2/A3 per-node topology and RoCE boundary"
    },
    {
      "url": "https://github.com/vllm-project/vllm-ascend/blob/2c8e27aba65c49c41e9ce43795d5db1643032097/docs/source/user_guide/support_matrix/supported_models.md",
      "sha256": "5260420ccc7ebde350f3fad37af0c0615b7ffddcc7ecd0ea89848f590643748a",
      "bytes": 21761,
      "role": "Pinned current support matrix used to confirm the experimental 200k row remains present"
    },
    {
      "url": "https://github.com/zai-org/GLM-5/blob/25206af860c4ac10f6411c597c574f9b1c00e53c/example/ascend.md",
      "sha256": "63b63aa23dcc7acc9eb41a93be4655da57e94c33750748c106193e493b515ec5",
      "bytes": 3135,
      "role": "Z.AI first-party Ascend deployment entry point and framework boundary"
    },
    {
      "url": "https://huggingface.co/zai-org/GLM-5.2-FP8/blob/3722e206c8461c750e5ab73f5528eba42d0c6a37/README.md",
      "sha256": "bb0d0f8616aabc567ab270d5d6f25efb8ac4d333eba9687657cee296719088cc",
      "bytes": 7488,
      "role": "Z.AI model card confirmation that Ascend NPU frameworks are supported"
    },
    {
      "url": "https://www.modelscope.cn/api/v1/models/ZhipuAI/GLM-5.2/repo/files?Revision=master&Root=",
      "sha256": "c106fa55db99de63d95f1183ef08af4d57dc2f73000a9410bb9ddb64f6b2ac5f",
      "bytes": 120151,
      "role": "Observed ModelScope BF16 public file manifest with per-file sizes, revisions, and SHA-256"
    },
    {
      "url": "https://www.modelscope.cn/api/v1/models/Eco-Tech/GLM-5.2-w8a8/repo/files?Revision=master&Root=",
      "sha256": "d5fe5eb5afa9d277181256e21b981a22b1a90d51a65f4e798f1c0d0a9393ea24",
      "bytes": 87672,
      "role": "Observed ModelScope W8A8 public file manifest with per-file sizes, revisions, and SHA-256"
    },
    {
      "url": "https://www.modelscope.cn/api/v1/models/Eco-Tech/GLM-5.2-w4a8c8/repo/files?Revision=master&Root=",
      "sha256": "cc05e25574ba0526c4cd703c26ebabeea51aac3dd55cd6c01d68c13396142414",
      "bytes": 47253,
      "role": "Observed ModelScope W4A8C8 public file manifest with per-file sizes, revisions, and SHA-256"
    },
    {
      "url": "https://www.modelscope.cn/models/Eco-Tech/GLM-5.2-w4a8c8/resolve/master/config.json",
      "sha256": "817f5fb39ca5d4c4b5648de89ca00deaea7537d8c2f130172a459252a05c1073",
      "bytes": 3699,
      "role": "Observed W4A8C8 architecture and native 1,048,576-position configuration"
    },
    {
      "url": "https://www.modelscope.cn/models/Eco-Tech/GLM-5.2-w4a8c8/resolve/master/GLM-5.2_best_practice.yaml",
      "sha256": "6ab02dd52174803873b2809b227be870d7acd9ed6846c5e55d093a3dc0acc72d",
      "bytes": 3342,
      "role": "Observed ModelSlim quantization scope and A2/A3 verification tags"
    },
    {
      "url": "https://api.github.com/repos/vllm-project/vllm-ascend/releases/latest",
      "sha256": "6fb2007fc30d1cb489f30bb03613765e7debb399644e1bd07d8937c43a2b478e",
      "bytes": 30360,
      "role": "Observed official v0.23.0 release contract, dependencies, GLM-5.2 1M scope, and known issues"
    },
    {
      "url": "https://api.github.com/repos/vllm-project/vllm-ascend/issues/12851",
      "sha256": "cd2d55725d9b5f2ccdace5dfc7eed7baa13bfb7dc324c855aa93e8195729b43b",
      "bytes": 4971,
      "role": "Closed stale report that v0.22.1rc1 failed to load the W4A8C8 artifact while v0.23.0rc1 reportedly started"
    },
    {
      "url": "https://api.github.com/repos/vllm-project/vllm-ascend/issues/12405",
      "sha256": "cf1b098ee361692b1a8fb61b396059b8e72861b9490f2866ea3b4a22d8c073b3",
      "bytes": 6366,
      "role": "Open triaged A2 W8A8 function-call no-output report"
    },
    {
      "url": "https://api.github.com/repos/vllm-project/vllm-ascend/issues/14378",
      "sha256": "59b5c60550518679495cb50b188f6f9b0e3505fcedcc3471fe9153d29ea25ef3",
      "bytes": 4274,
      "role": "Open triaged W4A8C8 PD plus SFA C8 repeated-accuracy fluctuation report"
    },
    {
      "url": "https://z.ai/blog/glm-5.2",
      "sha256": "63e94b9e11a2db64243ff18418011577f2a98ca7851ce9dcb0e97cfe34cc245a",
      "bytes": 598,
      "role": "Fixed per-round Z.AI GLM-5.2 release-page check"
    },
    {
      "url": "https://www.zhipuai.cn/zh/research",
      "sha256": "188772a5cb6b65eb65f02024d89d153b692ef3d3cb2c98ddd233352c0dd6ac80",
      "bytes": 1236461,
      "role": "Fixed per-round Zhipu research-index check"
    }
  ],
  "revisions": {
    "vllmAscendRelease": "v0.23.0",
    "vllmAscendReleaseCommit": "5cb98caaadeff42b5b62b996e34bb2aaa29d20fd",
    "vllmAscendReleasePublishedAt": "2026-08-16T22:18:14Z",
    "currentTutorialCommit": "b25f49dd0281726017041eed92bd0edc56961a2e",
    "currentSupportMatrixCommit": "2c8e27aba65c49c41e9ce43795d5db1643032097",
    "zaiAscendCommit": "25206af860c4ac10f6411c597c574f9b1c00e53c",
    "zaiFp8AscendCommit": "3722e206c8461c750e5ab73f5528eba42d0c6a37"
  },
  "artifactArithmetic": {
    "bf16": {
      "repository": "ZhipuAI/GLM-5.2",
      "primaryShardCount": 282,
      "primaryShardBytes": 1506667387408,
      "auxiliarySafetensorsBytes": 0,
      "manifestDigest": "a2e0c1027b1d166833f91eb92b7ab875549d4db5e03829a5e3178c9150b55fd1",
      "allShardHashesValid": true,
      "safetensorsBytes": 1506667387408,
      "decimalGB": 1506.667,
      "gibibytes": 1403.193,
      "shareOfOneA3NodePercent": 147.14,
      "evenSafetensorsGiBPer16Npus": 87.7,
      "evenSafetensorsGiBPer8Npus": 175.399
    },
    "w8a8": {
      "repository": "Eco-Tech/GLM-5.2-w8a8",
      "primaryShardCount": 181,
      "primaryShardBytes": 773800519384,
      "auxiliarySafetensorsBytes": 75497560,
      "manifestDigest": "cf768d8c6d6b41ce3d05b2ec88e59e7487df1cc7a5f1fd8bcd7ef5fd302c4b00",
      "allShardHashesValid": true,
      "safetensorsBytes": 773876016944,
      "decimalGB": 773.876,
      "gibibytes": 720.728,
      "shareOfOneA3NodePercent": 75.57,
      "evenSafetensorsGiBPer16Npus": 45.046,
      "evenSafetensorsGiBPer8Npus": 90.091
    },
    "w4a8c8": {
      "repository": "Eco-Tech/GLM-5.2-w4a8c8",
      "primaryShardCount": 95,
      "primaryShardBytes": 404718618504,
      "auxiliarySafetensorsBytes": 75497560,
      "manifestDigest": "f85f93044e962eaf3120f92a53d6739977a8d3ba36be0152e5aa3b76639f279d",
      "allShardHashesValid": true,
      "configMaxPositionEmbeddings": 1048576,
      "configQuantizationField": null,
      "requiresAscendQuantizationFlag": true,
      "safetensorsBytes": 404794116064,
      "decimalGB": 404.794,
      "gibibytes": 376.994,
      "shareOfOneA3NodePercent": 39.53,
      "evenSafetensorsGiBPer16Npus": 23.562,
      "evenSafetensorsGiBPer8Npus": 47.124
    },
    "reductions": {
      "w8a8FromBf16Percent": 48.64,
      "w4a8c8FromBf16Percent": 73.13,
      "w4a8c8FromW8a8Percent": 47.69
    }
  },
  "hardware": {
    "a3CurrentTutorial": {
      "nodes": 1,
      "npusPerNode": 16,
      "memoryPerNpuDecimalGB": 64,
      "aggregateDecimalGB": 1024
    },
    "a3ReleaseTutorialSingleNode": {
      "nodes": 1,
      "npusPerNode": 8,
      "memoryPerNpuDecimalGB": 128,
      "aggregateDecimalGB": 1024
    },
    "a2": {
      "nodes": 1,
      "npusPerNode": 8,
      "memoryPerNpuDecimalGB": 64,
      "aggregateDecimalGB": 512,
      "documentedMinimumW4Nodes": 2
    }
  },
  "scenarios": [
    {
      "id": "a3-single-below-1m",
      "hardware": "A3",
      "nodes": 1,
      "totalNpus": 16,
      "mode": "co-located",
      "parallelism": "DP2 TP8",
      "maxModelLen": 135000,
      "mtpDraftTokens": 3,
      "artifact": "w4a8c8"
    },
    {
      "id": "a3-dual-below-1m",
      "hardware": "A3",
      "nodes": 2,
      "totalNpus": 32,
      "mode": "co-located",
      "parallelism": "DP4 TP8",
      "maxModelLen": 66000,
      "mtpDraftTokens": 3,
      "artifact": "w4a8c8"
    },
    {
      "id": "a3-pd-below-1m",
      "hardware": "A3",
      "nodes": 4,
      "totalNpus": 64,
      "mode": "prefill-decode",
      "parallelism": "P DP4 TP8; D DP32 TP1",
      "maxModelLen": 133120,
      "mtpDraftTokens": "P1/D5",
      "artifact": "w4a8c8"
    },
    {
      "id": "a2-dual-below-1m",
      "hardware": "A2",
      "nodes": 2,
      "totalNpus": 16,
      "mode": "co-located",
      "parallelism": "DP2 TP8",
      "maxModelLen": 40000,
      "mtpDraftTokens": 5,
      "artifact": "w4a8c8"
    },
    {
      "id": "a2-pd-below-1m",
      "hardware": "A2",
      "nodes": 8,
      "totalNpus": 64,
      "mode": "prefill-decode",
      "parallelism": "P DP4 TP8; D DP8 TP4",
      "maxModelLen": 256000,
      "mtpDraftTokens": "P1/D3",
      "artifact": "w4a8c8"
    },
    {
      "id": "a3-single-1m",
      "hardware": "A3",
      "nodes": 1,
      "totalNpus": 16,
      "mode": "co-located",
      "parallelism": "DP1 TP16 DCP16",
      "maxModelLen": 1024000,
      "mtpDraftTokens": 3,
      "artifact": "w4a8c8"
    },
    {
      "id": "a3-dual-1m",
      "hardware": "A3",
      "nodes": 2,
      "totalNpus": 32,
      "mode": "co-located",
      "parallelism": "DP4 TP8 DCP8",
      "maxModelLen": 1024000,
      "mtpDraftTokens": 3,
      "artifact": "w4a8c8"
    },
    {
      "id": "a3-pd-1m",
      "hardware": "A3",
      "nodes": 4,
      "totalNpus": 64,
      "mode": "prefill-decode",
      "parallelism": "P/D DP4 TP8 DCP8",
      "maxModelLen": 1024000,
      "mtpDraftTokens": "P1/D3",
      "artifact": "w4a8c8"
    }
  ],
  "contextArithmetic": {
    "nativePositions": 1048576,
    "releaseSupportMatrixMax": 200000,
    "tutorialCommandMax": 1024000,
    "tutorialTuningTableMax": 1040000,
    "a2OneMillionValidated": false,
    "a3OneMillionValidated": true,
    "commandGapFromNativeTokens": 24576,
    "commandShareOfNativePercent": 97.65625,
    "tuningTableAboveCommandTokens": 16000,
    "tuningTableGapFromNativeTokens": 8576,
    "supportMatrixBelowCommandTokens": 824000
  },
  "documentDrift": {
    "releaseSingleA3": "128GB x 8",
    "currentSingleA3": "64GB x 16",
    "sameAggregateDecimalGB": 1024,
    "releaseBf16A2PerNodeText": "64GB x 32",
    "currentBf16A2PerNodeText": "64GB x 8",
    "rule": "Pin software to the stable release, but validate card count, memory, and command world size against current docs and observed hardware before launch."
  },
  "runtime": {
    "upstreamVllm": "v0.23.0",
    "python": ">=3.10,<3.13",
    "cann": "9.1.0",
    "pytorch": "2.10.0",
    "torchNpu": "2.10.0.post4",
    "tritonAscend": "3.2.2",
    "releaseImageA3": "quay.io/ascend/vllm-ascend:v0.23.0-a3",
    "releaseImageA2": "quay.io/ascend/vllm-ascend:v0.23.0"
  },
  "boundaries": {
    "supportMatrixStatus": "experimental",
    "w4PdSfaC8AccuracyReport": {
      "issue": 14378,
      "state": "open",
      "reportedRangePercentagePoints": "2-3",
      "recommendedAccuracySensitiveArtifact": "w8a8"
    },
    "a2W8ToolCallReport": {
      "issue": 12405,
      "state": "open",
      "status": "triaged",
      "symptom": "server generated tokens but the tool-call request returned no usable output"
    },
    "oldRcLoadReport": {
      "issue": 12851,
      "state": "closed",
      "stateReason": "not_planned",
      "reportedWorkingReplacement": "v0.23.0rc1-a3"
    },
    "dcpSfaC8TogetherRecommended": false,
    "pdDcpMustBeSymmetric": true,
    "oneMillionA2Validated": false,
    "languageModelEvaluationHarnessTested": false
  },
  "decisionRules": [
    "Use the stable v0.23.0 runtime family, but reconcile its versioned tutorial with the pinned current tutorial before launch because per-node A2/A3 card descriptions changed.",
    "Treat the 1,024,000-token commands as scenario-specific A3 evidence, not a blanket override of the experimental 200k support-matrix row or the native 1,048,576-position config.",
    "Keep enable_sparse_sfa_c8 off when DCP is enabled in v0.23.0; for P/D serving, DCP must be enabled on both sides or disabled on both sides.",
    "Use W4A8C8 only behind repeated task-quality gates. The open project issue recommends W8A8 for accuracy-sensitive production, but the tutorial does not provide an interchangeable W8 launch block.",
    "Shard bytes are a weight-file floor, not process fit. Verify every file hash, actual NPU count, per-rank peak memory, KV headroom, HCCL topology, parser behavior, and repeated outputs before promotion."
  ],
  "invariants": {
    "seventeenSourceReceiptsHashed": true,
    "noModelsOrAcceleratorsUsed": true,
    "artifactShardCountsMatch": true,
    "artifactHashesWerePresent": true,
    "eightScenarioRows": true,
    "threeOneMillionRowsAreA3Only": true,
    "oneMillionCommandIsBelowNative": true,
    "tuningTableDisagreesWithCommands": true,
    "releaseAndCurrentA3HaveSameAggregateButDifferentCardCounts": true,
    "w4PayloadIsBelowFortyPercentOfA3Aggregate": true,
    "w8PayloadIsAboveSeventyFivePercentOfA3Aggregate": true,
    "productionBoundariesFailClosed": true,
    "activeAccuracyReportsRemainVisible": true
  },
  "limits": [
    "ModelScope master manifests can change. The receipt freezes the observed response hash plus a deterministic digest over every safetensors filename, size, revision, and SHA-256.",
    "Published file bytes do not include runtime allocation, KV cache, graph capture, communication workspaces, allocator fragmentation, replicas, or failover reserve.",
    "Even per-NPU divisions are arithmetic only; expert, tensor, data, context, and pipeline placement can be uneven.",
    "Experimental support, verified tutorial commands, issue reports, and a successful local canary are different evidence levels. None proves production quality for a new workload.",
    "Open issue observations are project reports, not GLM52.ai reproductions. GLM52.ai made no model, NPU, or GPU call.",
    "The SerpAPI query used only for discovery and intent validation is recorded separately in the private operation log, not represented as a model-performance source."
  ]
}
