{
  "schemaVersion": 1,
  "operationId": "20260826042447-42217f9b8b",
  "checkedAt": "2026-08-26T04:52:00Z",
  "status": "pinned-source-zero-gpu-glm52-vllm-h200-runtime-oom-audit",
  "revisions": {
    "vllmLatestStableTag": "v0.27.1",
    "vllmLatestStableCommit": "6e448d0ea9bf3d88d898b65449ca6dc2aec170ac",
    "vllmMainCommit": "044b05220ee9eb42ae3b90b984d0c7e9eb442f0e",
    "flashMlaPinnedCommit": "a8f794d1251cbfd88a5011445dd5582289c727e4",
    "glm52Fp8Revision": "ba978f7d347eaf65d22f1a86833408afdb953541",
    "chunkingPr": 49357,
    "flashMlaFixPr": 19,
    "integrationPr": 53755
  },
  "sourceReceipts": [
    {
      "id": "zai-release",
      "url": "https://z.ai/blog/glm-5.2",
      "purpose": "fixed first-party release check and model context",
      "httpStatus": 200,
      "bytes": 598,
      "sha256": "63e94b9e11a2db64243ff18418011577f2a98ca7851ce9dcb0e97cfe34cc245a"
    },
    {
      "id": "zhipu-research",
      "url": "https://www.zhipuai.cn/zh/research",
      "purpose": "fixed first-party research-index check",
      "httpStatus": 200,
      "bytes": 1236461,
      "sha256": "188772a5cb6b65eb65f02024d89d153b692ef3d3cb2c98ddd233352c0dd6ac80"
    },
    {
      "id": "hf-model-card",
      "url": "https://huggingface.co/zai-org/GLM-5.2-FP8/raw/ba978f7d347eaf65d22f1a86833408afdb953541/README.md",
      "purpose": "pinned first-party checkpoint documentation",
      "httpStatus": 200,
      "bytes": 10909,
      "sha256": "de23c1b7cab43a99f0fedf4edf10de0d165882a2ecb2e2bad6c5796fcabf2e46"
    },
    {
      "id": "hf-config",
      "url": "https://huggingface.co/zai-org/GLM-5.2-FP8/raw/ba978f7d347eaf65d22f1a86833408afdb953541/config.json",
      "purpose": "pin GLM-5.2 architecture, query heads, index geometry, and context",
      "httpStatus": 200,
      "bytes": 29464,
      "sha256": "22e49334abf8562fecf70ca3292ba3f5b33f5602fb2bf10b52dd64a66cfe65ff"
    },
    {
      "id": "vllm-release-0271",
      "url": "https://api.github.com/repos/vllm-project/vllm/releases/tags/v0.27.1",
      "purpose": "pin the latest stable release at audit time",
      "httpStatus": 200,
      "bytes": 13964,
      "sha256": "0c6224baeffb5d951d0af0452d4ffc33a1a4dbb478c74c3c43bd143e495636e3"
    },
    {
      "id": "vllm-tag-flashmla-sparse",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/6e448d0ea9bf3d88d898b65449ca6dc2aec170ac/vllm/v1/attention/backends/mla/flashmla_sparse.py",
      "purpose": "audit the v0.27.1 mixed FP8 path, head padding, and unchunked full-token call",
      "httpStatus": 200,
      "bytes": 34530,
      "sha256": "0544c431bbf311d7283fee7f34cd7ad8a2df7961380646f2c12db2e861608ee4"
    },
    {
      "id": "vllm-tag-flashmla-pin",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/6e448d0ea9bf3d88d898b65449ca6dc2aec170ac/cmake/external_projects/flashmla.cmake",
      "purpose": "verify the FlashMLA source revision bundled by v0.27.1",
      "httpStatus": 200,
      "bytes": 8021,
      "sha256": "3cd4062759f9710f2f1ead18cc726605780b98cb3f584101e1fe9f5ab60e35cc"
    },
    {
      "id": "vllm-tag-scheduler",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/6e448d0ea9bf3d88d898b65449ca6dc2aec170ac/vllm/config/scheduler.py",
      "purpose": "pin max-num-batched-tokens and speculative scheduling semantics",
      "httpStatus": 200,
      "bytes": 12302,
      "sha256": "3cf5d41a5d662ab0eada6f3128e2538abc0a250d6aef82acab7df7dd84e498bc"
    },
    {
      "id": "vllm-tag-cache",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/6e448d0ea9bf3d88d898b65449ca6dc2aec170ac/vllm/config/cache.py",
      "purpose": "pin GPU memory utilization and KV cache configuration semantics",
      "httpStatus": 200,
      "bytes": 14663,
      "sha256": "0b603decec7487138e97b9cafbad9684ec22c8052f52a227236d1151a03aa34f"
    },
    {
      "id": "flashmla-pinned-source",
      "url": "https://raw.githubusercontent.com/vllm-project/FlashMLA/a8f794d1251cbfd88a5011445dd5582289c727e4/csrc/api/sparse_decode.h",
      "purpose": "reproduce the unconditional FP32 split-KV accumulator shapes",
      "httpStatus": 200,
      "bytes": 18739,
      "sha256": "9333a66b2a2d13583a72c4617ff76a5f361b1528da8d97b40d7ab26237a741e9"
    },
    {
      "id": "vllm-issue-53413",
      "url": "https://api.github.com/repos/vllm-project/vllm/issues/53413",
      "purpose": "capture the GLM-5.2 H200 delayed-OOM report and exact reproduction profile",
      "httpStatus": 200,
      "bytes": 8756,
      "sha256": "656a83676c2fe103c24597b48083b22dbf9a0998fc60e793882e9de25c0bc105"
    },
    {
      "id": "vllm-issue-53413-comments",
      "url": "https://api.github.com/repos/vllm-project/vllm/issues/53413/comments?per_page=100",
      "purpose": "capture maintainer workaround guidance and separately attributed operator evidence",
      "httpStatus": 200,
      "bytes": 5958,
      "sha256": "a210c7ca8265d8dc46a4575a07b4f255480d1d443b1043270dbf7b106a0cbb6e"
    },
    {
      "id": "vllm-pr-49357",
      "url": "https://api.github.com/repos/vllm-project/vllm/pulls/49357",
      "purpose": "audit the open mixed-batch chunking workaround and its throughput boundary",
      "httpStatus": 200,
      "bytes": 19486,
      "sha256": "7358a4cb1fe3746bb774d12ae5d43e7f9e50167e8e394ebcfc3789941347f159"
    },
    {
      "id": "vllm-pr-53755",
      "url": "https://api.github.com/repos/vllm-project/vllm/pulls/53755",
      "purpose": "audit the open FlashMLA integration fix and reported allocation test",
      "httpStatus": 200,
      "bytes": 21866,
      "sha256": "68626c42217640a8320494dd74d7bfda38c501d5ae57a489e1d11fdd72844aa8"
    },
    {
      "id": "flashmla-pr-19",
      "url": "https://api.github.com/repos/vllm-project/FlashMLA/pulls/19",
      "purpose": "audit the open upstream no-split workspace fix",
      "httpStatus": 200,
      "bytes": 16907,
      "sha256": "b2fcf55b6b19b91ebf066dc8686e00c7b760476ee343f50b3bb9bce6df038445"
    },
    {
      "id": "flashmla-pr-19-files",
      "url": "https://api.github.com/repos/vllm-project/FlashMLA/pulls/19/files?per_page=100",
      "purpose": "verify the proposed conditional allocation and regression test diff",
      "httpStatus": 200,
      "bytes": 8505,
      "sha256": "c30ed90955652c42c31e6f15c011b29e4454b5e0af34dc0b68685008c46258db"
    },
    {
      "id": "vllm-glm52-recipe",
      "url": "https://raw.githubusercontent.com/vllm-project/recipes/f65701647c2a123ba529dfebb52163c4947ffac9/models/zai-org/GLM-5.2.yaml",
      "purpose": "pin official GLM-5.2 serving guidance and model profile",
      "httpStatus": 200,
      "bytes": 15247,
      "sha256": "517662028453e8a0b62f53a8b507d6b8ac502aaecbd5452b396c37ea2fbae4a7"
    },
    {
      "id": "vllm-ascend-glm52",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm-ascend/b7de842253ad8da059c31ff7e5e9357974d7929a/docs/source/tutorials/models/GLM5.2.md",
      "purpose": "verify the documented MTP decode-budget relationship in a disaggregated GLM-5.2 profile",
      "httpStatus": 200,
      "bytes": 71812,
      "sha256": "c9ab1c9c885fb107d7a0e64d8553460bde3b3f2c9bd867eeeec6d84f1af7cb04"
    },
    {
      "id": "llmd-h200-guide",
      "url": "https://raw.githubusercontent.com/llm-d/llm-d/0a3dd07b4bde6406388c31355fccf8a1dc051780/guides/agentic-serving/glm-5-2-h200.md",
      "purpose": "verify that an official ecosystem H200 route uses prefill-decode disaggregation",
      "httpStatus": 200,
      "bytes": 13000,
      "sha256": "854d68148d80b12b068f08fbb277dc642ba8dac805f96714413df59ad9004a66"
    },
    {
      "id": "vllm-main-flashmla-pin",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/044b05220ee9eb42ae3b90b984d0c7e9eb442f0e/cmake/external_projects/flashmla.cmake",
      "purpose": "verify audited main still pins the affected FlashMLA revision",
      "httpStatus": 200,
      "bytes": 8021,
      "sha256": "3cd4062759f9710f2f1ead18cc726605780b98cb3f584101e1fe9f5ab60e35cc"
    },
    {
      "id": "vllm-main-flashmla-sparse",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/044b05220ee9eb42ae3b90b984d0c7e9eb442f0e/vllm/v1/attention/backends/mla/flashmla_sparse.py",
      "purpose": "verify audited main retains the full-token mixed FP8 call",
      "httpStatus": 200,
      "bytes": 39072,
      "sha256": "d1a4db2384f99cb748a00afcb46572a0425c9a43c2b2da9c5c032c6f3bd271ae"
    }
  ],
  "model": {
    "artifact": "zai-org/GLM-5.2-FP8",
    "architecture": "GlmMoeDsaForCausalLM",
    "modelType": "glm_moe_dsa",
    "decoderLayers": 78,
    "globalQueryHeads": 64,
    "indexTopK": 2048,
    "indexHeadDim": 128,
    "indexHeads": 32,
    "kvLoraRank": 512,
    "valueHeadDim": 256,
    "maxPositionEmbeddings": 1048576
  },
  "affectedProfile": {
    "hardware": "8x NVIDIA H200",
    "computeCapability": "SM90",
    "checkpoint": "zai-org/GLM-5.2-FP8",
    "vllmReportedVersion": "0.24.0",
    "tensorParallelSize": 8,
    "expertParallelEnabled": true,
    "kvCacheDtype": "fp8",
    "maxModelLength": 262144,
    "maxNumSequences": 32,
    "maxNumBatchedTokens": 32768,
    "mtpSpeculativeTokens": 5,
    "reportedRuntimeBeforeFailureHours": 61,
    "reportedFreeGiBBeforeFailedAllocation": 6.4,
    "reportedFailedAllocationGiB": 8,
    "reportedKvCacheUsagePercentRange": [
      2,
      5.4
    ],
    "boundary": "Issue-reporter evidence; GLM52.ai did not reproduce this GPU run."
  },
  "versionBoundary": [
    {
      "name": "vLLM v0.27.1",
      "commit": "6e448d0ea9bf3d88d898b65449ca6dc2aec170ac",
      "releasedAt": "2026-08-11T10:47:49Z",
      "flashMlaCommit": "a8f794d1251cbfd88a5011445dd5582289c727e4",
      "noSplitWorkspaceFixPresent": false,
      "decision": "affected source shape remains present"
    },
    {
      "name": "audited vLLM main",
      "commit": "044b05220ee9eb42ae3b90b984d0c7e9eb442f0e",
      "checkedAt": "2026-08-26T04:52:00Z",
      "flashMlaCommit": "a8f794d1251cbfd88a5011445dd5582289c727e4",
      "noSplitWorkspaceFixPresent": false,
      "decision": "affected FlashMLA pin and full-token mixed path remain present"
    },
    {
      "name": "open vLLM chunking PR 49357",
      "state": "open",
      "strategy": "bound mixed batches by chunking with an acknowledged throughput tradeoff",
      "decision": "source-evaluation candidate, not a production recommendation"
    },
    {
      "name": "open FlashMLA PR 19 plus vLLM PR 53755",
      "state": "open",
      "strategy": "skip unused FP32 split-KV accumulators when the scheduler has one SM partition",
      "reportedTestHardware": "one H100 80GB",
      "reportedPeakBeforeMiB": 42.3467,
      "reportedPeakAfterMiB": 8.2822,
      "decision": "wait for merge, pinned packaging, and H200 workload validation"
    }
  ],
  "workspaceFormula": {
    "totalSplits": 2,
    "accumulatorBytesPerToken": 262144,
    "accumulatorMiBPerToken": 0.25,
    "lseBytesPerToken": 512,
    "outputBytesPerToken": 65536,
    "dominantAccumulatorFormula": "(batch + SM partitions) x scheduled tokens x padded query heads x kernel value dimension x FP32 bytes",
    "boundary": "No-split shape matching the reported allocation and open upstream fix. Real allocations depend on the resolved kernel path, scheduler metadata, concurrency, graph pools, allocator state, and other workspaces."
  },
  "workspaceRows": [
    {
      "maxNumBatchedTokens": 4096,
      "dominantFp32AccumulatorGiB": 1,
      "lseAccumulatorMiB": 2,
      "bf16OutputGiB": 0.25,
      "dominantAccumulatorPlusOutputGiB": 1.25
    },
    {
      "maxNumBatchedTokens": 8192,
      "dominantFp32AccumulatorGiB": 2,
      "lseAccumulatorMiB": 4,
      "bf16OutputGiB": 0.5,
      "dominantAccumulatorPlusOutputGiB": 2.5
    },
    {
      "maxNumBatchedTokens": 16384,
      "dominantFp32AccumulatorGiB": 4,
      "lseAccumulatorMiB": 8,
      "bf16OutputGiB": 1,
      "dominantAccumulatorPlusOutputGiB": 5
    },
    {
      "maxNumBatchedTokens": 32768,
      "dominantFp32AccumulatorGiB": 8,
      "lseAccumulatorMiB": 16,
      "bf16OutputGiB": 2,
      "dominantAccumulatorPlusOutputGiB": 10
    },
    {
      "maxNumBatchedTokens": 65536,
      "dominantFp32AccumulatorGiB": 16,
      "lseAccumulatorMiB": 32,
      "bf16OutputGiB": 4,
      "dominantAccumulatorPlusOutputGiB": 20
    }
  ],
  "reportedFailureReconciliation": {
    "calculatedDominantFp32AccumulatorGiB": 8,
    "reportedFailedAllocationGiB": 8,
    "reportedFreeGiBBeforeFailedAllocation": 6.4,
    "reportedScratchDeficitGiB": 1.6,
    "mtpTokensPerSequencePerDecodeStep": 6,
    "decodeOnlyScheduledTokens": 192,
    "mixedBatchHeadroomTokens": 32576,
    "interpretation": "The pinned no-split accumulator shape reproduces the reported 8 GiB request. Decode-only MTP occupancy is small relative to the 32K cap; mixed prefill is required to approach the reported budget."
  },
  "mitigationGates": [
    "confirm the exact vLLM commit or immutable image digest and resolved FlashMLA source revision",
    "confirm GLM-5.2-FP8, FP8 sparse MLA, mixed prefill plus decode, and per-rank head padding select the affected path",
    "lower max-num-batched-tokens as the direct temporary control and test the resulting throughput and latency",
    "measure minimum free HBM and peak allocated and reserved bytes during a production-shaped mixed burst, not only startup KV cache usage",
    "disable MTP for the first control, then add five speculative tokens only after the plain mixed-batch arm passes",
    "consider prefill-decode disaggregation only when the extra hardware and transfer path are justified",
    "do not deploy an open PR or switch cache dtype without output-parity, failure, and rollback evidence",
    "promote only a pinned build containing the merged fix after soak, burst, cancellation, restart, and rollback tests"
  ],
  "runtimeBoundary": {
    "localGpuRuns": 0,
    "modelCalls": 0,
    "upstreamTestsExecutedLocally": 0,
    "issueReportsTreatedAsLocalBenchmarks": false,
    "openPullRequestsRecommendedForProduction": false,
    "generatedIllustrationIsBenchmark": false
  },
  "aiHot": {
    "fingerprint": "f1-b94ea55430da42a9",
    "batchArtifact": "tmp/aihot-seo-operations/batches/20260826042447-42217f9b8b.json",
    "batchSha256": "402a026f0b5294927c9be9ac1650ecbdf61ad04fd0e4cfbfb7bb560a258e596f",
    "classifications": [
      {
        "id": "cmt96t9tm0augrolyx2c3vi84",
        "permalink": "https://aihot.virxact.com/items/cmt96t9tm0augrolyx2c3vi84",
        "classification": "duplicate-adjacent-intent",
        "reason": "The LangChain and Airbyte ingestion item is adjacent to existing LangChain, LlamaIndex, and ColBERT retrieval pages but provides no direct new GLM-5.2 runtime decision."
      },
      {
        "id": "cmt90wf1p06j9roly5e34fmu3",
        "permalink": "https://aihot.virxact.com/items/cmt90wf1p06j9roly5e34fmu3",
        "classification": "weak-glm-link",
        "reason": "OpenWorker may accept local open models, but the discovery item did not verify a GLM-5.2 adapter, protocol, or reproducible task."
      },
      {
        "id": "cmt8z2eko055crolytaitdxv8",
        "permalink": "https://aihot.virxact.com/items/cmt8z2eko055crolytaitdxv8",
        "classification": "weak-glm-link",
        "reason": "Claude memory is a different vendor product and has no direct GLM-5.2 technical decision."
      },
      {
        "id": "cmt8uphwp3qmhro73czv8pbpv",
        "permalink": "https://aihot.virxact.com/items/cmt8uphwp3qmhro73czv8pbpv",
        "classification": "weak-glm-link",
        "reason": "WeatherNext cyclone prediction has no direct GLM-5.2 training, inference, or application boundary."
      },
      {
        "id": "cmt8vaqtw3r29ro73d67v03bc",
        "permalink": "https://aihot.virxact.com/items/cmt8vaqtw3r29ro73d67v03bc",
        "classification": "weak-glm-link",
        "reason": "The compute-supply forecast is broad commentary without a specific GLM-5.2 reader task or reproducible evidence block."
      },
      {
        "id": "cmt924nw007a4rolyhmqbswmh",
        "permalink": "https://aihot.virxact.com/items/cmt924nw007a4rolyhmqbswmh",
        "classification": "duplicate-intent",
        "reason": "Existing GLM-5.2 OpenRouter routing and API-provider pages already cover model and provider selection; another broad selector page would compete with them."
      },
      {
        "id": "cmt8zzh0w05nirolyozmvq7i0",
        "permalink": "https://aihot.virxact.com/items/cmt8zzh0w05nirolyozmvq7i0",
        "classification": "weak-glm-link",
        "reason": "A video-generation API is not a compatible use of the text-only GLM-5.2 checkpoint."
      }
    ]
  },
  "searchSupply": {
    "queries": [
      "GLM-5.2 vLLM OOM H200 sparse_decode_fwd",
      "GLM-5.2 max-num-batched-tokens OOM"
    ],
    "method": "public web search used only for result-supply discovery",
    "serpApiRequests": 0,
    "serpApiBoundary": "No paid request was made because the site-specific 250-request monthly ceiling was already exceeded in the shared ledger.",
    "finding": "The dedicated result was the upstream issue itself; adjacent results were generic deployment guides or different OOM paths. No result reconciled the latest stable pin, audited main pin, workspace arithmetic, temporary cap, and two open fixes."
  },
  "overlapAudit": {
    "prepublicationCanonicalCount": 90,
    "primaryIntent": "audit and bound the delayed GLM-5.2 FP8 FlashMLA sparse-decode workspace OOM on H200 under vLLM",
    "readerJob": "identify the affected path, calculate the batch-token workspace, choose a temporary cap or disaggregation gate, and wait for a merged pinned fix before production",
    "closestPages": [
      {
        "url": "/guides/glm-5-2-vllm-decode-context-parallelism/",
        "difference": "DCP version and correctness gates; it does not size the no-split FlashMLA accumulator or repair a delayed runtime OOM."
      },
      {
        "url": "/guides/glm-5-2-mtp-speculative-decoding/",
        "difference": "MTP configuration and acceptance benchmarking; it does not audit the mixed FP8 sparse workspace failure."
      },
      {
        "url": "/guides/run-glm-5-2-locally/",
        "difference": "Broad capacity and route selection; it does not diagnose an H200 service that starts successfully and later dies."
      },
      {
        "url": "/guides/glm-5-2-hicache-kv-offload/",
        "difference": "SGLang host and storage cache tiers; the reported vLLM failure is not KV-cache pressure."
      }
    ],
    "distinctIntent": true
  },
  "decision": {
    "classification": "affected-stable-and-main-use-bounded-temporary-control-wait-for-merged-fix",
    "stable": "Treat vLLM v0.27.1 as affected for the audited GLM-5.2 FP8 mixed sparse path because it pins the unfixed FlashMLA source.",
    "currentMain": "Treat audited main 044b0522 as affected because it retains the same FlashMLA pin and full-token mixed path.",
    "temporaryControl": "Reduce max-num-batched-tokens, disable MTP for the first control, and require a mixed prefill/decode burst plus soak test with measured HBM headroom. The 8192 value is a canary candidate, not a universal safe default.",
    "alternative": "Use prefill-decode disaggregation only when its extra GPUs, transfer path, and failure surface are justified and separately validated.",
    "fixBoundary": "Do not deploy either open PR as a production fix. Wait for merge and pinned packaging, then repeat correctness, allocation, service, failure, soak, and rollback tests."
  },
  "invariants": {
    "twentyOneSourceReceipts": true,
    "allSourcesReturned200": true,
    "allSourceHashesValid": true,
    "glm52ArchitecturePinned": true,
    "stablePinsAffectedFlashMla": true,
    "auditedMainPinsAffectedFlashMla": true,
    "bothFixPathsRemainOpen": true,
    "reportedEightGiBReproduced": true,
    "reportedPhysicalHeadroomInsufficient": true,
    "mtpDecodeOnlyBelowMixedTokenCap": true,
    "noLocalGpuOrModelRun": true,
    "openPrNotProductionRecommendation": true,
    "distinctIntent": true
  }
}
