{
  "schemaVersion": 1,
  "operationId": "20260829113457-17d55bae2c",
  "checkedAt": "2026-08-29T11:47:23.388Z",
  "status": "pinned-zero-runtime-glm52-vllm-torch-compile-support-audit",
  "sourceReceipts": [
    {
      "id": "zai-glm52-release",
      "url": "https://z.ai/blog/glm-5.2",
      "finalUrl": "https://z.ai/blog/glm-5.2",
      "purpose": "mandatory first-party GLM-5.2 release check",
      "httpStatus": 200,
      "bytes": 598,
      "sha256": "a9e8c2b6f34717d69e3a0aa26bb117256a4d8c95bd299910c2a693000ee88fe8"
    },
    {
      "id": "zhipu-research-index",
      "url": "http://zhipuai.cn/zh/research",
      "finalUrl": "https://www.zhipuai.cn/zh/research",
      "purpose": "mandatory Zhipu AI research-discovery check",
      "httpStatus": 200,
      "bytes": 1235119,
      "sha256": "7d50f4c290fbc240f50fabc8d18f7a499688f968f53951655572909ddf547d4e"
    },
    {
      "id": "hf-glm52-config",
      "url": "https://huggingface.co/zai-org/GLM-5.2-FP8/resolve/ba978f7d347eaf65d22f1a86833408afdb953541/config.json",
      "finalUrl": "https://huggingface.co/api/resolve-cache/models/zai-org/GLM-5.2-FP8/ba978f7d347eaf65d22f1a86833408afdb953541/config.json?%2Fzai-org%2FGLM-5.2-FP8%2Fresolve%2Fba978f7d347eaf65d22f1a86833408afdb953541%2Fconfig.json=&etag=%224e1f0168afd127189fb1c4ddb1d4476a4fca96ac%22",
      "purpose": "pin the GLM-5.2 architecture used by the runtime audit",
      "httpStatus": 200,
      "bytes": 29464,
      "sha256": "22e49334abf8562fecf70ca3292ba3f5b33f5602fb2bf10b52dd64a66cfe65ff"
    },
    {
      "id": "vllm-glm52-recipe",
      "url": "https://raw.githubusercontent.com/vllm-project/recipes/5943215a27acb4a243e9d27bd69daf491034cfea/models/zai-org/GLM-5.2.yaml",
      "finalUrl": "https://raw.githubusercontent.com/vllm-project/recipes/5943215a27acb4a243e9d27bd69daf491034cfea/models/zai-org/GLM-5.2.yaml",
      "purpose": "pin the official GLM-5.2 vLLM serving recipe",
      "httpStatus": 200,
      "bytes": 15247,
      "sha256": "517662028453e8a0b62f53a8b507d6b8ac502aaecbd5452b396c37ea2fbae4a7"
    },
    {
      "id": "vllm-issue-54197",
      "url": "https://api.github.com/repos/vllm-project/vllm/issues/54197",
      "finalUrl": "https://api.github.com/repos/vllm-project/vllm/issues/54197",
      "purpose": "capture the open torch.compile support report",
      "httpStatus": 200,
      "bytes": 5644,
      "sha256": "3a063d1124de758e25327367572b2bc3e31ce9960cba7d340997e9cdac9ec5bd"
    },
    {
      "id": "vllm-issue-54197-comments",
      "url": "https://api.github.com/repos/vllm-project/vllm/issues/54197/comments",
      "finalUrl": "https://api.github.com/repos/vllm-project/vllm/issues/54197/comments",
      "purpose": "capture maintainer-facing root-cause follow-up",
      "httpStatus": 200,
      "bytes": 6068,
      "sha256": "06eb3fa5337b1cdfa6f72904f28c949e97e165880483825e23d20e10bfb9a30b"
    },
    {
      "id": "vllm-stable-commit",
      "url": "https://api.github.com/repos/vllm-project/vllm/commits/v0.28.0",
      "finalUrl": "https://api.github.com/repos/vllm-project/vllm/commits/v0.28.0",
      "purpose": "resolve the v0.28.0 stable tag",
      "httpStatus": 200,
      "bytes": 4868,
      "sha256": "07376e59bec891c8edfeb322579383c523f352d524a5a043eb56a5c67bb1bac9"
    },
    {
      "id": "vllm-issue-commit",
      "url": "https://api.github.com/repos/vllm-project/vllm/commits/4a6a3272e8d75518efe0a6f9393eb504f3ed2ee0",
      "finalUrl": "https://api.github.com/repos/vllm-project/vllm/commits/4a6a3272e8d75518efe0a6f9393eb504f3ed2ee0",
      "purpose": "resolve the issue reporter source snapshot",
      "httpStatus": 200,
      "bytes": 13224,
      "sha256": "937bf7d4323dee11ee961b6b916e61ddaf21dee628efeb40f57afdc04bfacdab"
    },
    {
      "id": "vllm-checked-main-commit",
      "url": "https://api.github.com/repos/vllm-project/vllm/commits/cacc429f62c3738c9c95093e9bd410e96103221a",
      "finalUrl": "https://api.github.com/repos/vllm-project/vllm/commits/cacc429f62c3738c9c95093e9bd410e96103221a",
      "purpose": "pin the checked main-branch snapshot",
      "httpStatus": 200,
      "bytes": 22606,
      "sha256": "896506262b4749c8e72add14ddbdac113d0b72c69432a21bb7b5680171d5d47c"
    },
    {
      "id": "vllm-direct-pr-search",
      "url": "https://api.github.com/search/issues?q=repo%3Avllm-project%2Fvllm+54197+in%3Abody+is%3Apr&per_page=20",
      "finalUrl": "https://api.github.com/search/issues?q=repo%3Avllm-project%2Fvllm+54197+in%3Abody+is%3Apr&per_page=20",
      "purpose": "check for a pull request directly referencing issue 54197",
      "httpStatus": 200,
      "bytes": 79,
      "sha256": "c9938edecb99d754b2d039ac9eec320a94c769b6a6417921e2dcbef9e2fe01a0"
    },
    {
      "id": "stable-model",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/2cf0a6915ce544dc493a0990f2ea38d81601128a/vllm/models/deepseek_v32/nvidia/model.py",
      "finalUrl": "https://raw.githubusercontent.com/vllm-project/vllm/2cf0a6915ce544dc493a0990f2ea38d81601128a/vllm/models/deepseek_v32/nvidia/model.py",
      "purpose": "check the stable GLM model class for compile decoration",
      "httpStatus": 200,
      "bytes": 16571,
      "sha256": "e3b329da4d411ae9ef54043a3a664dc5defef1271a9aae4ff77267165a9e8574"
    },
    {
      "id": "issue-model",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/4a6a3272e8d75518efe0a6f9393eb504f3ed2ee0/vllm/models/deepseek_v32/nvidia/model.py",
      "finalUrl": "https://raw.githubusercontent.com/vllm-project/vllm/4a6a3272e8d75518efe0a6f9393eb504f3ed2ee0/vllm/models/deepseek_v32/nvidia/model.py",
      "purpose": "check the issue snapshot GLM model class for compile decoration",
      "httpStatus": 200,
      "bytes": 17923,
      "sha256": "93bd68fb8c2eab2351b1e141cd6a22e944a78276b0c7f657ce96ed50ced750fa"
    },
    {
      "id": "checked-main-model",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/cacc429f62c3738c9c95093e9bd410e96103221a/vllm/models/deepseek_v32/nvidia/model.py",
      "finalUrl": "https://raw.githubusercontent.com/vllm-project/vllm/cacc429f62c3738c9c95093e9bd410e96103221a/vllm/models/deepseek_v32/nvidia/model.py",
      "purpose": "check the pinned main snapshot GLM model class for compile decoration",
      "httpStatus": 200,
      "bytes": 17923,
      "sha256": "93bd68fb8c2eab2351b1e141cd6a22e944a78276b0c7f657ce96ed50ced750fa"
    },
    {
      "id": "stable-vllm-config",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/2cf0a6915ce544dc493a0990f2ea38d81601128a/vllm/config/vllm.py",
      "finalUrl": "https://raw.githubusercontent.com/vllm-project/vllm/2cf0a6915ce544dc493a0990f2ea38d81601128a/vllm/config/vllm.py",
      "purpose": "check stable model-runner routing and unsupported-compile warning",
      "httpStatus": 200,
      "bytes": 118577,
      "sha256": "fb07f95658ae0ee6e6c16c0f36e90f9179e8761221a45d4bf1aebe8bcf63cdcb"
    },
    {
      "id": "issue-vllm-config",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/4a6a3272e8d75518efe0a6f9393eb504f3ed2ee0/vllm/config/vllm.py",
      "finalUrl": "https://raw.githubusercontent.com/vllm-project/vllm/4a6a3272e8d75518efe0a6f9393eb504f3ed2ee0/vllm/config/vllm.py",
      "purpose": "check issue-snapshot model-runner routing and unsupported-compile warning",
      "httpStatus": 200,
      "bytes": 125199,
      "sha256": "367af6c12e5dbdb60ed3abe863ce26e28be3ebb8fb0c4eb69a666d572573eb8e"
    },
    {
      "id": "issue-compilation-config",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/4a6a3272e8d75518efe0a6f9393eb504f3ed2ee0/vllm/config/compilation.py",
      "finalUrl": "https://raw.githubusercontent.com/vllm-project/vllm/4a6a3272e8d75518efe0a6f9393eb504f3ed2ee0/vllm/config/compilation.py",
      "purpose": "pin compilation mode semantics",
      "httpStatus": 200,
      "bytes": 68576,
      "sha256": "6e57a553093753856baaf7987e37ff24a15b73c95730e397e74d22c539d440ec"
    },
    {
      "id": "issue-compile-wrapper",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/4a6a3272e8d75518efe0a6f9393eb504f3ed2ee0/vllm/compilation/wrapper.py",
      "finalUrl": "https://raw.githubusercontent.com/vllm-project/vllm/4a6a3272e8d75518efe0a6f9393eb504f3ed2ee0/vllm/compilation/wrapper.py",
      "purpose": "pin the fullgraph torch.compile wrapper call",
      "httpStatus": 200,
      "bytes": 14270,
      "sha256": "c1b1fca679ea16aa07a696831a810c0d531da2bb0ea32dd6f1a95cdaef36de07"
    },
    {
      "id": "issue-compile-decorators",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/4a6a3272e8d75518efe0a6f9393eb504f3ed2ee0/vllm/compilation/decorators.py",
      "finalUrl": "https://raw.githubusercontent.com/vllm-project/vllm/4a6a3272e8d75518efe0a6f9393eb504f3ed2ee0/vllm/compilation/decorators.py",
      "purpose": "pin the compile-support decorator contract",
      "httpStatus": 200,
      "bytes": 32039,
      "sha256": "e2231c1de4ee3a89f37eccae8cdab1f4842b824c163c5f9eeed9c16bd15dd925"
    },
    {
      "id": "issue-fused-q",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/4a6a3272e8d75518efe0a6f9393eb504f3ed2ee0/vllm/models/deepseek_v32/nvidia/ops/fused_q_cutedsl.py",
      "finalUrl": "https://raw.githubusercontent.com/vllm-project/vllm/4a6a3272e8d75518efe0a6f9393eb504f3ed2ee0/vllm/models/deepseek_v32/nvidia/ops/fused_q_cutedsl.py",
      "purpose": "pin the runtime platform-capability check named in the report",
      "httpStatus": 200,
      "bytes": 18294,
      "sha256": "548992905e056e4021e052609664f422c753f24a3eaaac46963957442c2ba808"
    },
    {
      "id": "issue-workspace",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/4a6a3272e8d75518efe0a6f9393eb504f3ed2ee0/vllm/v1/worker/workspace.py",
      "finalUrl": "https://raw.githubusercontent.com/vllm-project/vllm/4a6a3272e8d75518efe0a6f9393eb504f3ed2ee0/vllm/v1/worker/workspace.py",
      "purpose": "pin the ContextVar workspace path named in the report",
      "httpStatus": 200,
      "bytes": 11181,
      "sha256": "d21c08167d1d0d1c0cddaa80453a53a0b4dec5d81b5a537c94a21324344976b3"
    },
    {
      "id": "issue-torch-compile-design",
      "url": "https://raw.githubusercontent.com/vllm-project/vllm/4a6a3272e8d75518efe0a6f9393eb504f3ed2ee0/docs/design/torch_compile.md",
      "finalUrl": "https://raw.githubusercontent.com/vllm-project/vllm/4a6a3272e8d75518efe0a6f9393eb504f3ed2ee0/docs/design/torch_compile.md",
      "purpose": "pin vLLM torch.compile design expectations",
      "httpStatus": 200,
      "bytes": 19600,
      "sha256": "45cb6b8664e9926172541a79cd2be2f6686ae0c7f585ec65f9de64796c88aa2b"
    }
  ],
  "pins": {
    "modelRevision": "ba978f7d347eaf65d22f1a86833408afdb953541",
    "stableVllmTag": "v0.28.0",
    "stableVllmRevision": "2cf0a6915ce544dc493a0990f2ea38d81601128a",
    "issueVllmRevision": "4a6a3272e8d75518efe0a6f9393eb504f3ed2ee0",
    "checkedMainRevision": "cacc429f62c3738c9c95093e9bd410e96103221a",
    "recipeRevision": "5943215a27acb4a243e9d27bd69daf491034cfea"
  },
  "modelContract": {
    "architectures": [
      "GlmMoeDsaForCausalLM"
    ],
    "modelType": "glm_moe_dsa",
    "hiddenLayers": 78,
    "routedExperts": 256,
    "activeExpertsPerToken": 8,
    "indexTopK": 2048,
    "transformersVersion": "5.12.0",
    "quantizationMethod": "fp8"
  },
  "activationAudit": {
    "stableV0280": {
      "tag": "v0.28.0",
      "resolvedSha": "2cf0a6915ce544dc493a0990f2ea38d81601128a",
      "committedAt": "2026-08-24T23:42:42Z",
      "modelClassPresent": true,
      "modelClassCompileDecorated": false,
      "glmDefaultsToV2Runner": false,
      "unsupportedCompileWarningPresent": true
    },
    "issueSnapshot": {
      "resolvedSha": "4a6a3272e8d75518efe0a6f9393eb504f3ed2ee0",
      "committedAt": "2026-08-27T11:06:20Z",
      "modelClassPresent": true,
      "modelClassCompileDecorated": false,
      "glmDefaultsToV2Runner": true,
      "unsupportedCompileWarningPresent": true
    },
    "checkedMainSnapshot": {
      "resolvedSha": "cacc429f62c3738c9c95093e9bd410e96103221a",
      "committedAt": "2026-08-29T06:52:23Z",
      "modelClassPresent": true,
      "modelClassCompileDecorated": false
    },
    "allCheckedModelClassesLackCompileDecorator": true,
    "runnerBoundaryChangedAfterStable": true,
    "stableAndIssueWarnOnUnsupportedCompile": true,
    "modeThreeIsVllmCompile": true,
    "v1DefaultsToModeThree": true,
    "fullgraphWrapper": true,
    "sourceHazardsMatchReport": true,
    "officialRecipeAddsExplicitModeThree": false
  },
  "issueStatus": {
    "number": 54197,
    "title": "[Bug] GlmMoeDsa (V2 model runner) cannot use torch.compile — silent fallback, then fullgraph failures when force-enabled",
    "url": "https://github.com/vllm-project/vllm/issues/54197",
    "state": "open",
    "createdAt": "2026-08-28T10:06:49Z",
    "updatedAt": "2026-08-29T07:54:35Z",
    "labels": [
      "torch.compile"
    ],
    "commentCount": 2,
    "directPullRequestsFound": 0,
    "environmentCommitNamed": true,
    "modeThreeNamed": true,
    "unsupportedWarningObserved": true,
    "eagerFallbackReported": true,
    "forceDecoratorFullgraphFailuresReported": true,
    "v1A100FailureReported": true,
    "maintainerFacingRootCauseFollowup": true,
    "prDescribedAsReadyButNotSubmitted": true
  },
  "decision": {
    "currentVerdict": "Do not treat GLM-5.2 mode 3 as a compiled performance profile on the pinned v0.28.0, issue, or checked-main snapshots.",
    "safeAction": "Fail startup on the unsupported-model warning, retain a known-good unmodified runtime, and wait for a merged pinned fix before an eager-versus-compiled parity and performance canary.",
    "unsafeAction": "Do not add @support_torch_compile locally and infer success from process health or an accepted flag.",
    "evidenceBoundary": "Static source and issue-state evidence can prove support markers and reported failures, but it cannot prove compiled performance, output parity, GPU compatibility, or a future fix."
  },
  "canaryGate": {
    "algorithm": [
      "Reject any startup log containing the model-does-not-support-torch.compile warning.",
      "Require a positive non-zero compilation receipt from the exact process and pinned source.",
      "Require deterministic output, tool-call, long-context, and failure-path parity against the known-good profile.",
      "Only then run a bounded latency and throughput canary with immediate rollback."
    ],
    "fixtures": [
      {
        "name": "reported-glm52-fallback",
        "unsupportedWarning": true,
        "compilationSeconds": 0,
        "parityPassed": false,
        "decision": "reject",
        "reason": "unsupported-model-warning"
      },
      {
        "name": "quiet-noop",
        "unsupportedWarning": false,
        "compilationSeconds": 0,
        "parityPassed": false,
        "decision": "reject",
        "reason": "no-positive-compilation-receipt"
      },
      {
        "name": "compiled-without-parity",
        "unsupportedWarning": false,
        "compilationSeconds": 38.2,
        "parityPassed": false,
        "decision": "hold",
        "reason": "compiled-output-parity-not-proven"
      },
      {
        "name": "compiled-and-parity-checked",
        "unsupportedWarning": false,
        "compilationSeconds": 38.2,
        "parityPassed": true,
        "decision": "eligible-for-bounded-canary",
        "reason": "activation-and-parity-receipts-present"
      }
    ]
  },
  "overlapAudit": {
    "prepublicationSitemapUrls": 96,
    "registryPages": 89,
    "distinctIntent": true,
    "primaryIntent": "audit whether GLM-5.2 actually enters vLLM torch.compile and detect an unsupported eager fallback before a performance canary",
    "readerJob": "pin the runtime source, inspect the compile-support marker and startup receipts, reject silent eager fallback, avoid unsafe local patching, and rerun parity after an upstream fix",
    "nearby": [
      [
        "https://glm52.ai/guides/glm-5-2-vllm-sequence-parallel-moe/",
        "Owns TP/DP/EP sequence-parallel topology and a one-token state-guard failure, not compile activation."
      ],
      [
        "https://glm52.ai/guides/glm-5-2-vllm-thinking-token-budget/",
        "Owns reasoning-budget capacity and parser transitions, not torch.compile support."
      ],
      [
        "https://glm52.ai/guides/glm-5-2-vllm-offline-batch-inference/",
        "Owns JSONL validation, execution, reconciliation, and quarantine, not graph compilation."
      ],
      [
        "https://glm52.ai/guides/run-glm-5-2-locally/",
        "Owns hardware and checkpoint fit, not a pinned vLLM compiler capability audit."
      ]
    ]
  },
  "searchSupply": {
    "query": "GLM-5.2 vLLM torch.compile",
    "serpApiRequests": 1,
    "requestId": "serpapi-d946b778443c42bd93e7221062d9b232",
    "knownMonthlyUsageAfter": 821,
    "result": "http-error",
    "evaluation": "failed",
    "retryCount": 0,
    "decisionImpact": "No Google evidence was returned. The topic, wording, overlap boundary, and claims remain based on pinned primary source and the local registry."
  },
  "aiHot": [
    {
      "itemId": "cmte36nzj02isrog2hwxpthhn",
      "permalink": "https://aihot.virxact.com/items/cmte36nzj02isrog2hwxpthhn",
      "classification": "weak-glm-link",
      "note": "Autonomous mathematical discovery supplies no GLM-5.2 serving or compiler contract."
    },
    {
      "itemId": "cmte242e701jdrog2vz9p2cy4",
      "permalink": "https://aihot.virxact.com/items/cmte242e701jdrog2vz9p2cy4",
      "classification": "weak-glm-link",
      "note": "A Qwen local-performance report concerns another model family and cannot establish GLM-5.2 compile support."
    },
    {
      "itemId": "cmtdxtxi809gyro2m2zykqzli",
      "permalink": "https://aihot.virxact.com/items/cmtdxtxi809gyro2m2zykqzli",
      "classification": "duplicate-event",
      "note": "The GLM-5.3 weight-release event is already covered by the comparison, API migration, and checkpoint migration canonicals."
    },
    {
      "itemId": "cmtdqqeiu046gro2maw35y3c4",
      "permalink": "https://aihot.virxact.com/items/cmtdqqeiu046gro2maw35y3c4",
      "classification": "weak-glm-link",
      "note": "The Cursor/OpenAI access report has no direct GLM-5.2 runtime or compiler task."
    },
    {
      "itemId": "cmtdqc5oj03wzro2mgyc5a49x",
      "permalink": "https://aihot.virxact.com/items/cmtdqc5oj03wzro2mgyc5a49x",
      "classification": "duplicate-event",
      "note": "This is a second source for the same Cursor/OpenAI event and still has no direct GLM-5.2 task."
    }
  ],
  "discoveryChecks": {
    "zaiGlm52Release": "fetched and hashed with HTTP 200",
    "zhipuResearchIndex": "fetched and hashed after its redirect with non-empty bytes"
  },
  "protectedBaseline": {
    "homepageSha256": "5d7f8327af1af910c0484cdd0b7fde8695a107339c207cc585452210f7fd05a8",
    "backlinkPageSha256": "e4bf36153e8ab178f432bf672345581aea1e1a9cd9483c644df9b8179cd8212b",
    "homepageTitle": "GLM-5.2 Developer Guide: API, Pricing & Tools | GLM52.ai",
    "backlinkTitle": "GLM-5.2 vLLM Sequence-Parallel MoE Audit | GLM52.ai"
  },
  "runtimeBoundary": {
    "modelCalls": 0,
    "modelWeightBytesDownloaded": 0,
    "vllmImports": 0,
    "localGpuRuns": 0,
    "servingProcesses": 0,
    "containers": 0,
    "note": "Static source and issue-state evidence can prove support markers and reported failures, but it cannot prove compiled performance, output parity, GPU compatibility, or a future fix."
  },
  "invariants": {
    "twentyOneSourceReceipts": true,
    "allSourcesReturned200": true,
    "allSourceHashesValid": true,
    "immutablePinsMatch": true,
    "glmArchitecturePinned": true,
    "allCheckedSnapshotsLackDecorator": true,
    "runnerAndWarningPathsPinned": true,
    "compileContractPinned": true,
    "issueOpenWithoutDirectPr": true,
    "reportAndSourceMechanismsAlign": true,
    "fourGateFixturesCoverAllDecisions": true,
    "distinctIntent": true,
    "oneFailedSerpRequestNoRetry": true,
    "fiveAiHotItemsClassified": true,
    "zeroRuntimeExecution": true
  }
}
