{
  "schemaVersion": 1,
  "operationId": "20260825003437-353e8d76f8",
  "checkedAt": "2026-08-25T01:26:11Z",
  "status": "pinned-source-zero-runtime-vllm-dcp-version-and-correctness-gate",
  "revisions": {
    "vllmStableTag": "v0.26.0",
    "vllmStableCommit": "568afb3a13806beb53bb2e6bd518269357b237c0",
    "vllmMainCommit": "09fddeb4ce7ee8cd0c46f1da9c45aafb9a17e89a",
    "flashMlaDcpMergeCommit": "63ff748f657af0e2a95a712fd2cc994f4299ff8c",
    "openCorrectnessPr": 50005,
    "openCorrectnessPrHead": "6127a179f89384df0b0128584794a64dcaa8578b",
    "recipeCommit": "c3cb50c4d95c0be93bee03e7ea4e0740a0777158",
    "glm52Fp8Revision": "ba978f7d347eaf65d22f1a86833408afdb953541"
  },
  "sourceReceipts": [
    {
      "id": "zai-release",
      "url": "https://z.ai/blog/glm-5.2",
      "bytes": 598,
      "sha256": "63e94b9e11a2db64243ff18418011577f2a98ca7851ce9dcb0e97cfe34cc245a"
    },
    {
      "id": "zhipu-research",
      "url": "https://www.zhipuai.cn/zh/research",
      "bytes": 1236461,
      "sha256": "188772a5cb6b65eb65f02024d89d153b692ef3d3cb2c98ddd233352c0dd6ac80"
    },
    {
      "id": "vllm-dcp-docs",
      "url": "https://docs.vllm.ai/en/latest/serving/context_parallel_deployment/",
      "bytes": 604758,
      "sha256": "4e3c9420833ccf64edb7872e34daa006eddf77e0642828d63cf85b944a394493"
    },
    {
      "id": "vllm-dcp-blog",
      "url": "https://vllm.ai/blog/2026-08-07-decode-context-parallelism",
      "bytes": 144138,
      "sha256": "3f0457fff3b3509d98350baf3f5ce1b4845e3e208e26d6c3a8a91f1c338024c8"
    },
    {
      "id": "vllm-v026-release-api",
      "url": "https://api.github.com/repos/vllm-project/vllm/releases/tags/v0.26.0",
      "bytes": 32286,
      "sha256": "89874ee5e0afb0be8b4602b2159268366786a4a9271ef915215fa1ab1f25d373"
    },
    {
      "id": "vllm-issue-53134-api",
      "url": "https://api.github.com/repos/vllm-project/vllm/issues/53134",
      "bytes": 7933,
      "sha256": "9f4078daa18900fe322a77ed5eb0ad98d460c628c5cdf7e5e24797c7f7002a4e"
    },
    {
      "id": "vllm-pr-46514-api",
      "url": "https://api.github.com/repos/vllm-project/vllm/pulls/46514",
      "bytes": 37206,
      "sha256": "55d8268e04bfbfafc8ee682181212ebaedc4c9d65adf91a46d3529d45b4b2468"
    },
    {
      "id": "vllm-pr-46514-files-api",
      "url": "https://api.github.com/repos/vllm-project/vllm/pulls/46514/files?per_page=100",
      "bytes": 22943,
      "sha256": "82d0e8d11798c907ed5364ff063dcb7432f1287f21d3196c113d34493e1795bd"
    },
    {
      "id": "vllm-pr-46514-merge-api",
      "url": "https://api.github.com/repos/vllm-project/vllm/commits/63ff748f657af0e2a95a712fd2cc994f4299ff8c",
      "bytes": 27787,
      "sha256": "88dafaf0280787b4abce7517fa7b6e86310b5c78de04adc04ce4eb893404ad81"
    },
    {
      "id": "vllm-v026-commit-api",
      "url": "https://api.github.com/repos/vllm-project/vllm/commits/568afb3a13806beb53bb2e6bd518269357b237c0",
      "bytes": 4312,
      "sha256": "7dfdd57e8a5d461e6fb8e7f1b52b0b923e18f419ccf5c6c79931f3d163024458"
    },
    {
      "id": "vllm-main-commit-api",
      "url": "https://api.github.com/repos/vllm-project/vllm/commits/09fddeb4ce7ee8cd0c46f1da9c45aafb9a17e89a",
      "bytes": 5748,
      "sha256": "e76ccefe84e20b5efea9d900cfa591318c4108595efeecc1cd00c5a527b7e3d6"
    },
    {
      "id": "vllm-v026-flashmla-sparse",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/568afb3a13806beb53bb2e6bd518269357b237c0/vllm/v1/attention/backends/mla/flashmla_sparse.py",
      "bytes": 34530,
      "sha256": "0544c431bbf311d7283fee7f34cd7ad8a2df7961380646f2c12db2e861608ee4"
    },
    {
      "id": "vllm-main-flashmla-sparse",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/09fddeb4ce7ee8cd0c46f1da9c45aafb9a17e89a/vllm/v1/attention/backends/mla/flashmla_sparse.py",
      "bytes": 39072,
      "sha256": "d1a4db2384f99cb748a00afcb46572a0425c9a43c2b2da9c5c032c6f3bd271ae"
    },
    {
      "id": "vllm-v026-cp-utils",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/568afb3a13806beb53bb2e6bd518269357b237c0/vllm/v1/worker/cp_utils.py",
      "bytes": 10922,
      "sha256": "6e519640429d8c5a147db7c6b1df7c6fb1f5e22c1c86f525aea15622e1d47c14"
    },
    {
      "id": "vllm-main-sparse-mla-tests",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/09fddeb4ce7ee8cd0c46f1da9c45aafb9a17e89a/tests/v1/attention/test_sparse_mla_backends.py",
      "bytes": 60313,
      "sha256": "51eb2dd2d75bc0212c1b1661c99afad7e6b62324b8fdd6d5954c19af8a774adb"
    },
    {
      "id": "vllm-v026-flashinfer-sparse",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/568afb3a13806beb53bb2e6bd518269357b237c0/vllm/v1/attention/backends/mla/flashinfer_mla_sparse.py",
      "bytes": 16907,
      "sha256": "550079022e858952c37bfa1198dbb1c77884571af30eb4c2f1a8055cd6ca7b07"
    },
    {
      "id": "vllm-glm52-recipe",
      "url": "https://raw.githubusercontent.com/vllm-project/recipes/c3cb50c4d95c0be93bee03e7ea4e0740a0777158/models/zai-org/GLM-5.2.yaml",
      "bytes": 15247,
      "sha256": "517662028453e8a0b62f53a8b507d6b8ac502aaecbd5452b396c37ea2fbae4a7"
    },
    {
      "id": "glm52-fp8-config",
      "url": "https://huggingface.co/zai-org/GLM-5.2-FP8/resolve/ba978f7d347eaf65d22f1a86833408afdb953541/config.json",
      "bytes": 29464,
      "sha256": "22e49334abf8562fecf70ca3292ba3f5b33f5602fb2bf10b52dd64a66cfe65ff"
    },
    {
      "id": "glm52-fp8-model-card",
      "url": "https://huggingface.co/zai-org/GLM-5.2-FP8/resolve/ba978f7d347eaf65d22f1a86833408afdb953541/README.md",
      "bytes": 10909,
      "sha256": "de23c1b7cab43a99f0fedf4edf10de0d165882a2ecb2e2bad6c5796fcabf2e46"
    },
    {
      "id": "llmd-glm52-h200-blog",
      "url": "https://llm-d.ai/blog/serving-glm-5-2-agentic-workloads-on-llm-d",
      "bytes": 115376,
      "sha256": "2049aabec4c5c0eb76c3299913c461c965751e03b674360343a37a9f0808ef87"
    },
    {
      "id": "vllm-issue-50095-api",
      "url": "https://api.github.com/repos/vllm-project/vllm/issues/50095",
      "bytes": 5335,
      "sha256": "29c35b5ef4d14087f4d877f1897427c22606ab5e66025d12139ba2888d240db5"
    },
    {
      "id": "vllm-pr-50005-api",
      "url": "https://api.github.com/repos/vllm-project/vllm/pulls/50005",
      "bytes": 29773,
      "sha256": "53287fdd392be37c80cdafd5208faa35f7bc093574b86799b8e2d157d27bf5a2"
    },
    {
      "id": "vllm-pr-50005-files-api",
      "url": "https://api.github.com/repos/vllm-project/vllm/pulls/50005/files?per_page=100",
      "bytes": 9487,
      "sha256": "ec195c13b4e2d24b2d3a495dad44ad6db79f84c6930721f69649e7a20a6cd3f4"
    },
    {
      "id": "vllm-main-glm-attention",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/09fddeb4ce7ee8cd0c46f1da9c45aafb9a17e89a/vllm/models/deepseek_v32/attention.py",
      "bytes": 20783,
      "sha256": "fcd05acadabccd2ced25a42bbbde8c78e99699c3aac0556780bf145567913ad2"
    },
    {
      "id": "vllm-main-glm-kernels",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/09fddeb4ce7ee8cd0c46f1da9c45aafb9a17e89a/vllm/models/deepseek_v32/common/kernels.py",
      "bytes": 37515,
      "sha256": "0b520b15dc82c026b05f73ac8dba713549dcea6fa400f945543595c5038228ed"
    }
  ],
  "model": {
    "artifact": "zai-org/GLM-5.2-FP8",
    "architecture": "GlmMoeDsaForCausalLM",
    "modelType": "glm_moe_dsa",
    "decoderLayers": 78,
    "globalQueryHeads": 64,
    "effectiveKvHeads": 1,
    "kvLoraRank": 512,
    "qkRopeHeadDim": 64,
    "qkNopeHeadDim": 192,
    "valueHeadDim": 256,
    "indexTopK": 2048,
    "maxPositionEmbeddings": 1048576
  },
  "targetProfile": {
    "hardware": "8x NVIDIA H200",
    "computeCapability": "SM90",
    "checkpoint": "zai-org/GLM-5.2-FP8",
    "tensorParallelSize": 8,
    "expertParallelEnabled": true,
    "candidateDecodeContextParallelSizes": [
      1,
      2,
      4,
      8
    ],
    "kvCacheDtype": "fp8_ds_mla",
    "maxModelLength": 131072,
    "control": "same pinned image and checkpoint with decode_context_parallel_size=1"
  },
  "releaseBoundary": [
    {
      "name": "vLLM v0.26.0",
      "commit": "568afb3a13806beb53bb2e6bd518269357b237c0",
      "publishedAt": "2026-07-27T01:06:58Z",
      "flashMlaSparseReturnsDecodeLse": false,
      "pureDcpCorrectnessFixPresent": false,
      "decision": "reject DCP for this profile; the compatibility check fails during initialization"
    },
    {
      "name": "FlashMLA sparse DCP backend merge",
      "commit": "63ff748f657af0e2a95a712fd2cc994f4299ff8c",
      "mergedAt": "2026-08-19T04:02:10Z",
      "flashMlaSparseReturnsDecodeLse": true,
      "pureDcpCorrectnessFixPresent": false,
      "decision": "backend capability landed after v0.26.0, but it is not the full GLM-5.2 correctness boundary"
    },
    {
      "name": "audited vLLM main",
      "commit": "09fddeb4ce7ee8cd0c46f1da9c45aafb9a17e89a",
      "checkedAt": "2026-08-25T01:26:11Z",
      "flashMlaSparseReturnsDecodeLse": true,
      "pureDcpCorrectnessFixPresent": false,
      "decision": "do not promote; open PR 50005 still carries the fused GLM-5.2 pure-DCP correctness fix"
    },
    {
      "name": "open correctness PR 50005",
      "commit": "6127a179f89384df0b0128584794a64dcaa8578b",
      "updatedAt": "2026-08-24T21:22:58Z",
      "state": "open",
      "flashMlaSparseReturnsDecodeLse": true,
      "pureDcpCorrectnessFixPresent": true,
      "decision": "isolated source-evaluation candidate only; wait for merge, pinned release packaging, and an independent canary before production"
    }
  ],
  "stableToBackendMergeDays": 23.1217,
  "backendCapabilityMatrix": [
    {
      "build": "v0.26.0",
      "backend": "FLASHMLA_SPARSE",
      "h200Sm90": true,
      "fp8DsMla": true,
      "decodeLse": false,
      "glm52PureDcpCorrectnessFix": false,
      "result": "startup-blocked"
    },
    {
      "build": "v0.26.0",
      "backend": "FLASHINFER_MLA_SPARSE SM10",
      "h200Sm90": false,
      "fp8DsMla": false,
      "decodeLse": true,
      "glm52PureDcpCorrectnessFix": false,
      "result": "backend-selection-blocked"
    },
    {
      "build": "v0.26.0",
      "backend": "FLASHINFER_MLA_SPARSE_SM120",
      "h200Sm90": false,
      "fp8DsMla": true,
      "decodeLse": true,
      "glm52PureDcpCorrectnessFix": false,
      "result": "blackwell-only-not-h200"
    },
    {
      "build": "main at 09fddeb4",
      "backend": "FLASHMLA_SPARSE",
      "h200Sm90": true,
      "fp8DsMla": true,
      "decodeLse": true,
      "glm52PureDcpCorrectnessFix": false,
      "result": "initializes-but-production-blocked-by-open-correctness-fix"
    },
    {
      "build": "open PR 50005 head on a compatible post-46514 base",
      "backend": "FLASHMLA_SPARSE",
      "h200Sm90": true,
      "fp8DsMla": true,
      "decodeLse": true,
      "glm52PureDcpCorrectnessFix": true,
      "result": "isolated-canary-only"
    }
  ],
  "headEnvelopeRows": [
    {
      "dcpSize": 1,
      "localQueryHeads": 8,
      "gatheredQueryHeads": 8,
      "localPaddedHeads": 64,
      "gatheredPaddedHeads": 64,
      "backendHeadEnvelopeAccepted": true
    },
    {
      "dcpSize": 2,
      "localQueryHeads": 8,
      "gatheredQueryHeads": 16,
      "localPaddedHeads": 64,
      "gatheredPaddedHeads": 64,
      "backendHeadEnvelopeAccepted": true
    },
    {
      "dcpSize": 4,
      "localQueryHeads": 8,
      "gatheredQueryHeads": 32,
      "localPaddedHeads": 64,
      "gatheredPaddedHeads": 64,
      "backendHeadEnvelopeAccepted": true
    },
    {
      "dcpSize": 8,
      "localQueryHeads": 8,
      "gatheredQueryHeads": 64,
      "localPaddedHeads": 64,
      "gatheredPaddedHeads": 64,
      "backendHeadEnvelopeAccepted": true
    }
  ],
  "kvDuplicationRows": [
    {
      "dcpSize": 1,
      "totalKvCopiesAcrossTpGroup": 8,
      "sequenceFractionStoredPerRank": 1,
      "theoreticalDuplicationReductionPercentVsDcp1": 0
    },
    {
      "dcpSize": 2,
      "totalKvCopiesAcrossTpGroup": 4,
      "sequenceFractionStoredPerRank": 0.5,
      "theoreticalDuplicationReductionPercentVsDcp1": 50
    },
    {
      "dcpSize": 4,
      "totalKvCopiesAcrossTpGroup": 2,
      "sequenceFractionStoredPerRank": 0.25,
      "theoreticalDuplicationReductionPercentVsDcp1": 75
    },
    {
      "dcpSize": 8,
      "totalKvCopiesAcrossTpGroup": 1,
      "sequenceFractionStoredPerRank": 0.125,
      "theoreticalDuplicationReductionPercentVsDcp1": 87.5
    }
  ],
  "reportedIssueCell": {
    "source": "vLLM issue 53134",
    "build": "vLLM 0.26.0",
    "hardware": "8xH200 SM90",
    "checkpoint": "GLM-5.2-FP8",
    "tensorParallelSize": 8,
    "decodeContextParallelSize": 8,
    "maxRequestTokens": 262144,
    "reportedKvTokensPerGpu": 393000,
    "reportedAvailableKvCacheGb": 20,
    "reportedMaximumConcurrency": 1.5,
    "boundary": "Issue-reporter observation in an air-gapped deployment; not reproduced by GLM52.ai.",
    "calculatedConcurrencyFromRoundedKvTokens": 1.4992
  },
  "upstreamValidation": {
    "backendPr46514": {
      "hardware": "4xH200",
      "checkpoint": "GLM-5.2-NVFP4",
      "topology": "TP4/DCP4/EP with MTP",
      "testsPassed": 140,
      "testsFailed": 0,
      "needleContextTokens": 198000,
      "warmBatch1TokensPerSecond": 78,
      "batch32AggregateTokensPerSecond": 854,
      "dcpParity": "DCP1/2/4 byte-identical on the named long-context cases",
      "boundary": "Pull-request author measurements and tests; different checkpoint and topology from the 8xH200 FP8 target."
    },
    "correctnessPr50005": {
      "hardware": "one GB200 for the focused suite plus independent 4xH200 validation",
      "checkpoint": "GLM-5.2-NVFP4 for the multi-GPU validation",
      "topology": "TP4/DCP4, fp8_ds_mla, greedy decoding",
      "focusedTestsPassed": 67,
      "patchedResult": "tested DCP4 output byte-identical to DCP1",
      "unpatchedResult": "repeated-token output reported on the tested request",
      "boundary": "Open pull-request evidence; not a merged release and not reproduced by GLM52.ai."
    },
    "generalDcpBlog": {
      "hardware": "NVIDIA B200",
      "checkpoint": "Kimi K2.6 NVFP4",
      "baselinePlateauTokensPerSecondPerGpu": 1863,
      "dcpC512TokensPerSecondPerGpu": 6091,
      "dcpC512KvUsagePercent": 82,
      "boundary": "General DCP benchmark only; it must not be transferred to GLM-5.2, H200, FP8, or this topology."
    }
  },
  "canaryGates": [
    "pin a vLLM commit or immutable image digest that contains both the backend support and the eventual merged correctness fix",
    "prove resolved FLASHMLA_SPARSE, SM90, fp8_ds_mla, ag_rs, mixed-batch FP8, and an accepted head envelope",
    "start with DCP1 control and MTP disabled, then compare one DCP size at a time",
    "require deterministic output parity across short, long, mixed-batch, empty-local-shard, tool-call, and stop-condition cases",
    "record usable KV tokens, total HBM, admission limit, TTFT, TPOT, accepted output throughput, and communication time",
    "add MTP and full CUDA graphs only after the plain decode arm passes",
    "rollback to DCP1 with the same image, checkpoint, parsers, sampling profile, and request corpus"
  ],
  "runtimeBoundary": {
    "localGpuRuns": 0,
    "modelCalls": 0,
    "upstreamTestsExecutedLocally": 0,
    "productionRecommendationFromOpenPr": false,
    "generatedIllustrationIsBenchmark": false
  },
  "aiHot": {
    "fingerprint": "f1-4d4e7c691f35eb15",
    "batchArtifact": "tmp/aihot-seo-operations/batches/20260825003437-353e8d76f8.json",
    "batchSha256": "846377ac4e06e62663274e92eb178405ca242fd590e8858cc26142871e020e9e",
    "classifications": [
      {
        "id": "cmt7nq1d02bs1ro7373u88po4",
        "permalink": "https://aihot.virxact.com/items/cmt7nq1d02bs1ro7373u88po4",
        "classification": "weak-glm-link",
        "reason": "MetaRoCE is a new transport protocol, but no primary source tied it to the pinned GLM-5.2 vLLM path."
      },
      {
        "id": "cmt7e0g7y232wro738qg82x67",
        "permalink": "https://aihot.virxact.com/items/cmt7e0g7y232wro738qg82x67",
        "classification": "weak-glm-link",
        "reason": "Vera Rubin NVL72 is future vendor hardware evidence without a validated GLM-5.2 deployment route."
      },
      {
        "id": "cmt7e3m7h234gro73wwwhymb0",
        "permalink": "https://aihot.virxact.com/items/cmt7e3m7h234gro73wwwhymb0",
        "classification": "weak-glm-link",
        "reason": "The OpenAI agents story has no direct GLM-5.2 technical decision."
      },
      {
        "id": "cmt7nzq6i2c27ro73tqxagz44",
        "permalink": "https://aihot.virxact.com/items/cmt7nzq6i2c27ro73tqxagz44",
        "classification": "duplicate-intent",
        "reason": "The site already covers GPT-5.6 comparison and Kiro-compatible agent decisions; another news page would add little GLM-5.2 evidence."
      }
    ],
    "selectedRoute": "mandatory-proactive-research",
    "selectedTopic": "GLM-5.2 vLLM decode context parallelism version and correctness gate on H200",
    "directSources": [
      "https://github.com/vllm-project/vllm/issues/53134",
      "https://github.com/vllm-project/vllm/pull/46514",
      "https://github.com/vllm-project/vllm/issues/50095",
      "https://github.com/vllm-project/vllm/pull/50005"
    ]
  },
  "serp": {
    "requestId": "serpapi-450ec7285259454cabb49956fb9559b3",
    "query": "GLM-5.2 vLLM decode context parallel H200",
    "engine": "google",
    "locale": "United States / English",
    "requestedResults": 10,
    "requestCost": 1,
    "value": "high-value",
    "exactIntentResultFound": false,
    "decisionImpact": "The result set surfaced official DCP, GLM recipe, issue, and llm-d sources but no page that separated v0.26.0, the later backend merge, and the still-open GLM correctness fix."
  },
  "overlapAudit": {
    "prepublicationSitemapCanonicals": 88,
    "exactDecodeContextParallelPageFound": false,
    "closestPages": [
      "https://glm52.ai/guides/glm-5-2-vllm-sequence-parallel-moe/",
      "https://glm52.ai/guides/glm-5-2-dsa-cache-layer-split/",
      "https://glm52.ai/guides/glm-5-2-hicache-kv-offload/",
      "https://glm52.ai/guides/run-glm-5-2-locally/"
    ],
    "distinctIntent": true,
    "distinction": "This page gates sequence-sharded sparse-MLA decode KV ownership by exact vLLM backend and correctness revisions; the nearby pages cover MoE token-axis sequence parallelism, SGLang prefill cache ownership, KV offload, and general capacity."
  },
  "decision": {
    "classification": "reject-stable-and-current-main-wait-for-merged-correctness-fix",
    "stable": "Do not add DCP to vLLM v0.26.0 for GLM-5.2 FP8 on H200.",
    "currentMain": "Do not promote audited main: backend LSE support is present, but the GLM-5.2 pure-DCP fused-attention correctness fix is still open.",
    "futureCanary": "After the correctness fix merges and is pinned in a release or immutable image, compare DCP1 with one DCP size at a time under deterministic output, memory, service, failure, and rollback gates.",
    "rollback": "Return to decode_context_parallel_size=1 in the identical pinned image and checkpoint; do not substitute a different attention backend on H200 to force startup."
  },
  "invariants": {
    "twentyFiveSourceReceipts": true,
    "allSourceHashesValid": true,
    "glm52ArchitecturePinned": true,
    "h200Tp8ProfilePinned": true,
    "stableLacksDecodeLse": true,
    "backendFixLandedAfterStable": true,
    "auditedMainHasBackendLse": true,
    "auditedMainLacksCorrectnessFix": true,
    "correctnessPrStillOpen": true,
    "glmTp8Dcp8PassesOnlyHeadEnvelope": true,
    "dcp8ReducesRelativeCopiesToOne": true,
    "reporterConcurrencyArithmeticMatchesRoundedClaim": true,
    "noLocalGpuOrModelRun": true,
    "distinctIntent": true
  }
}
