{
  "schemaVersion": 1,
  "operationId": "20260822061545-7cc6dd6133",
  "checkedAt": "2026-08-22T14:29:34+08:00",
  "status": "pinned-source-zero-runtime-weight-cache-compatibility-audit",
  "scope": {
    "authenticatedApiCalls": 0,
    "liveModelCalls": 0,
    "graderCalls": 0,
    "gpuRuns": 0,
    "checkpointDownloads": 0,
    "credentialsUsed": false,
    "description": "Pinned public-source and runtime-source compatibility audit with deterministic arithmetic only; no GLM, MiniMax, GPU, checkpoint shard, or serving endpoint was used."
  },
  "discovery": {
    "aiHotItemId": "cmt3wt4qm13utro6t0hb6ga3y",
    "aiHotPermalink": "https://aihot.virxact.com/items/cmt3wt4qm13utro6t0hb6ga3y",
    "aiHotRole": "Untrusted discovery lead for the SGLang Weight Cache Daemon announcement; all technical claims were rechecked against first-party blog, issue, release, source, tests, and the pinned GLM-5.2-FP8 manifest."
  },
  "sourceReceipts": [
    {
      "url": "https://api.github.com/repos/sgl-project/sglang/releases/latest",
      "sha256": "c54c2ce5a5c72ce248b6c31902d983595445ec80745625eb4a363de27861cb7e",
      "bytes": 53100,
      "role": "Observed SGLang v0.5.18 release tag and publication time"
    },
    {
      "url": "https://api.github.com/repos/sgl-project/sglang/commits/v0.5.18",
      "sha256": "8241eec1d54eea1101e9c713c396de5c10fbed9fa30bc44895545deb6f1f4ff8",
      "bytes": 12417,
      "role": "Resolved v0.5.18 to immutable source commit 71de97b264b04dcd514cf904003028aefe9775c8"
    },
    {
      "url": "https://github.com/sgl-project/sglang/blob/71de97b264b04dcd514cf904003028aefe9775c8/python/sglang/srt/weight_cache/protocol.py",
      "sha256": "e687631db8e443a5da97b1d6c05ab6a1cadc84e327a6c4c4fb1ea3080befde96",
      "bytes": 14117,
      "role": "Pinned cache fingerprint, socket namespace, environment stamp, and quantization allowlist"
    },
    {
      "url": "https://github.com/sgl-project/sglang/blob/71de97b264b04dcd514cf904003028aefe9775c8/python/sglang/srt/server_args.py",
      "sha256": "5ed5badef8dca0ee3dd2b34a1c9c417e172ff1a947d9fe802f7983fb4a2be950",
      "bytes": 444424,
      "role": "Pinned off, daemon, and client semantics plus the speculative-decoding rejection"
    },
    {
      "url": "https://github.com/sgl-project/sglang/blob/71de97b264b04dcd514cf904003028aefe9775c8/python/sglang/srt/weight_cache/ipc_loader.py",
      "sha256": "a916a9d33fb9a02abe3693575473223497aac418d4718fb84e86396388ebeacc",
      "bytes": 23971,
      "role": "Pinned client fallback, hard-error, zero-copy mapping, and daemon-liveness behavior"
    },
    {
      "url": "https://github.com/sgl-project/sglang/blob/71de97b264b04dcd514cf904003028aefe9775c8/python/sglang/srt/weight_cache/daemon.py",
      "sha256": "d27b31e30c5acd81604fc245c748ba5d175ab0b2ccf0da5fc2bc5c3e7576182b",
      "bytes": 36884,
      "role": "Pinned daemon launch, rank calculation, and readiness behavior"
    },
    {
      "url": "https://github.com/sgl-project/sglang/blob/71de97b264b04dcd514cf904003028aefe9775c8/test/registered/model_loading/test_weight_cache_daemon.py",
      "sha256": "5a55a77111d561b3b79c91c411039ab6feaa579cfafe9193182035bb9e0ff966",
      "bytes": 12775,
      "role": "Pinned registered Weight Cache Daemon tests using Qwen3-0.6B at TP1 and TP2"
    },
    {
      "url": "https://github.com/sgl-project/sglang/blob/71de97b264b04dcd514cf904003028aefe9775c8/test/manual/test_weight_cache_e2e.py",
      "sha256": "42701bd77c0c609119e2547456d8ad392f9668fbc0f6bb1f1e3ee4f9a5f3aba8",
      "bytes": 12530,
      "role": "Pinned manual generic Weight Cache Daemon end-to-end harness"
    },
    {
      "url": "https://github.com/sgl-project/sglang/blob/71de97b264b04dcd514cf904003028aefe9775c8/test/registered/unit/model_loader/test_weight_cache_protocol.py",
      "sha256": "8af708cd89f7ae010dd487a77e092c32e08ff8ae57fb1eb85b8f5d0d0d2693b2",
      "bytes": 13241,
      "role": "Pinned protocol and allowlist unit tests"
    },
    {
      "url": "https://github.com/sgl-project/sglang/blob/71de97b264b04dcd514cf904003028aefe9775c8/docs/cookbook/autoregressive/GLM/GLM-5.2.mdx",
      "sha256": "ed999a20d87297acaf62a4f5c14fd0e3fec054077e71594edb630f54fcc92b11",
      "bytes": 20574,
      "role": "Pinned GLM-5.2 checkpoint, precision, MTP, hardware, and deployment guidance"
    },
    {
      "url": "https://github.com/sgl-project/sglang/blob/71de97b264b04dcd514cf904003028aefe9775c8/test/registered/8-gpu-models/test_glm52_fp8.py",
      "sha256": "510193a236f9495202e967402a11f14bded99422e3ede812dd43bad02b2edcd3",
      "bytes": 2102,
      "role": "Pinned GLM-5.2-FP8 TP8, TP8 plus DP8, and TP8 plus DP8 plus MTP tests, none combined with the Weight Cache Daemon"
    },
    {
      "url": "https://huggingface.co/api/models/zai-org/GLM-5.2-FP8",
      "sha256": "bbd3cbd146bada36de4607569930646d9947c82883256fcd36a1e7af6b259bfa",
      "bytes": 35757,
      "role": "Observed public GLM-5.2-FP8 revision, parameter dtypes, and repository storage"
    },
    {
      "url": "https://huggingface.co/zai-org/GLM-5.2-FP8/resolve/ba978f7d347eaf65d22f1a86833408afdb953541/config.json",
      "sha256": "22e49334abf8562fecf70ca3292ba3f5b33f5602fb2bf10b52dd64a66cfe65ff",
      "bytes": 29464,
      "role": "Pinned GLM-5.2-FP8 architecture and 128 by 128 block-FP8 quantization configuration"
    },
    {
      "url": "https://huggingface.co/zai-org/GLM-5.2-FP8/resolve/ba978f7d347eaf65d22f1a86833408afdb953541/model.safetensors.index.json",
      "sha256": "e0fe7f28c1f853d4824e4d796374e3dacf1fe470988773952c79b063768134bf",
      "bytes": 11359251,
      "role": "Pinned GLM-5.2-FP8 tensor-to-shard index and exact published tensor byte total"
    },
    {
      "url": "https://github.com/sgl-project/sglang/issues/33522",
      "sha256": "df74f9ed9f8bacbe9b3fc81cbb66a0798e9dcfe6d30a6d18e7e12a9a187e1120",
      "bytes": 13983,
      "role": "Observed open roadmap identifying DP, EP, node-rank, namespace, lifecycle, metrics, and security gaps"
    },
    {
      "url": "https://github.com/sgl-project/sglang/pull/27139",
      "sha256": "cfbe5b377af3c1b05a7e1f1430aa7626fc8883422e830245222f441f81e20565",
      "bytes": 55349,
      "role": "Observed merged Phase 1 implementation and immutable merge commit"
    },
    {
      "url": "https://www.lmsys.org/blog/2026-08-21-sglang-fast-recovery/",
      "sha256": "dc0ab29060ee9c5354bbeb5d7754bb3536367d4535434340650e4b68a73156ff",
      "bytes": 85401,
      "role": "First-party architecture description and Ling-2.6-1T publisher measurements"
    },
    {
      "url": "https://z.ai/blog/glm-5.2",
      "sha256": "63e94b9e11a2db64243ff18418011577f2a98ca7851ce9dcb0e97cfe34cc245a",
      "bytes": 598,
      "role": "Fixed per-round Z.AI GLM-5.2 release-page check"
    },
    {
      "url": "https://www.zhipuai.cn/zh/research",
      "sha256": "188772a5cb6b65eb65f02024d89d153b692ef3d3cb2c98ddd233352c0dd6ac80",
      "bytes": 1236461,
      "role": "Fixed per-round Zhipu research-index check"
    }
  ],
  "revisions": {
    "sglangRelease": "v0.5.18",
    "sglangReleasePublishedAt": "2026-08-22T00:09:15Z",
    "sglangSourceCommit": "71de97b264b04dcd514cf904003028aefe9775c8",
    "weightCacheMergeCommit": "f9c14e6bd4e17653ed2727c94010aa02ee84d794",
    "weightCacheMergedAt": "2026-07-25T06:31:22Z",
    "roadmapIssueState": "open",
    "roadmapIssueUpdatedAt": "2026-08-22T04:04:50Z",
    "glm52Fp8": "ba978f7d347eaf65d22f1a86833408afdb953541",
    "glm52Fp8LastModified": "2026-07-02T08:09:27.000Z"
  },
  "model": {
    "repository": "zai-org/GLM-5.2-FP8",
    "architecture": "GlmMoeDsaForCausalLM",
    "modelType": "glm_moe_dsa",
    "hiddenLayers": 78,
    "routedExperts": 256,
    "expertsPerToken": 8,
    "nextTokenPredictionLayers": 1,
    "quantMethod": "fp8",
    "quantFormat": "e4m3",
    "weightBlockSize": [
      128,
      128
    ],
    "activationScheme": "dynamic",
    "publishedTensorBytes": 755617140416,
    "tensorIndexEntries": 118629,
    "safetensorShards": 141,
    "reportedParameters": 753375793584,
    "reportedFp8Parameters": 751226191872,
    "reportedBf16Parameters": 2103729152,
    "reportedF32Parameters": 45872560,
    "publishedTensorGiB": 703.723301,
    "fp8ParameterSharePercent": 99.7147,
    "averageIndexEntriesPerShard": 841.34
  },
  "weightCache": {
    "modes": [
      "off",
      "daemon",
      "client"
    ],
    "standaloneRequiredForRestartAcceleration": true,
    "defaultTimeoutSeconds": 1800,
    "socketTemplate": "/tmp/sglang_weight_cache_rank{global_rank}.sock",
    "globalRankFormula": "tp_size * pp_rank + tp_rank",
    "cacheConfigFields": [
      "model_path",
      "model_arch",
      "tp_size",
      "tp_rank",
      "pp_size",
      "pp_rank",
      "dp_size",
      "ep_size",
      "quant_method",
      "quant_config_hash",
      "dtype",
      "revision",
      "device_capability",
      "torch_version"
    ],
    "missingShardIdentityFields": [
      "dp_rank",
      "ep_rank",
      "node_rank"
    ],
    "allowlist": {
      "unquantized": true,
      "blockWiseFp8": true,
      "perTensorFp8": false,
      "nvfp4": false,
      "mxfp4": false
    },
    "speculativeCombinationRejected": true,
    "clientMissingSocketFallsBackToDisk": true,
    "clientConfigMismatchHardErrors": true,
    "clientConnectionRefusedHardErrors": true,
    "daemonAnyCacheFailureHardErrors": true,
    "daemonDeathTerminatesAttachedClient": true,
    "engineSpawnedDaemonPersistsAcrossRestart": false,
    "blogSaysClientMismatchFallsBack": true,
    "sourceSaysClientMismatchHardErrors": true
  },
  "fingerprintAudit": {
    "declaredFieldCount": 14,
    "declaredFields": [
      "model_path",
      "model_arch",
      "tp_size",
      "tp_rank",
      "pp_size",
      "pp_rank",
      "dp_size",
      "ep_size",
      "quant_method",
      "quant_config_hash",
      "dtype",
      "revision",
      "device_capability",
      "torch_version"
    ],
    "missingShardIdentityFields": [
      "dp_rank",
      "ep_rank",
      "node_rank"
    ],
    "sizeFieldsWithoutCorrespondingRanks": [
      "dp_size",
      "ep_size"
    ],
    "socketAxes": [
      "tp_rank",
      "pp_rank"
    ],
    "absentSocketAxes": [
      "dp_rank",
      "ep_rank",
      "node_rank",
      "instance_namespace",
      "physical_gpu_uuid"
    ]
  },
  "topologyExamples": {
    "tp8Pp1": [
      {
        "ppRank": 0,
        "tpRank": 0,
        "globalRank": 0,
        "socket": "/tmp/sglang_weight_cache_rank0.sock"
      },
      {
        "ppRank": 0,
        "tpRank": 1,
        "globalRank": 1,
        "socket": "/tmp/sglang_weight_cache_rank1.sock"
      },
      {
        "ppRank": 0,
        "tpRank": 2,
        "globalRank": 2,
        "socket": "/tmp/sglang_weight_cache_rank2.sock"
      },
      {
        "ppRank": 0,
        "tpRank": 3,
        "globalRank": 3,
        "socket": "/tmp/sglang_weight_cache_rank3.sock"
      },
      {
        "ppRank": 0,
        "tpRank": 4,
        "globalRank": 4,
        "socket": "/tmp/sglang_weight_cache_rank4.sock"
      },
      {
        "ppRank": 0,
        "tpRank": 5,
        "globalRank": 5,
        "socket": "/tmp/sglang_weight_cache_rank5.sock"
      },
      {
        "ppRank": 0,
        "tpRank": 6,
        "globalRank": 6,
        "socket": "/tmp/sglang_weight_cache_rank6.sock"
      },
      {
        "ppRank": 0,
        "tpRank": 7,
        "globalRank": 7,
        "socket": "/tmp/sglang_weight_cache_rank7.sock"
      }
    ],
    "tp8Pp2": [
      {
        "ppRank": 0,
        "tpRank": 0,
        "globalRank": 0,
        "socket": "/tmp/sglang_weight_cache_rank0.sock"
      },
      {
        "ppRank": 0,
        "tpRank": 1,
        "globalRank": 1,
        "socket": "/tmp/sglang_weight_cache_rank1.sock"
      },
      {
        "ppRank": 0,
        "tpRank": 2,
        "globalRank": 2,
        "socket": "/tmp/sglang_weight_cache_rank2.sock"
      },
      {
        "ppRank": 0,
        "tpRank": 3,
        "globalRank": 3,
        "socket": "/tmp/sglang_weight_cache_rank3.sock"
      },
      {
        "ppRank": 0,
        "tpRank": 4,
        "globalRank": 4,
        "socket": "/tmp/sglang_weight_cache_rank4.sock"
      },
      {
        "ppRank": 0,
        "tpRank": 5,
        "globalRank": 5,
        "socket": "/tmp/sglang_weight_cache_rank5.sock"
      },
      {
        "ppRank": 0,
        "tpRank": 6,
        "globalRank": 6,
        "socket": "/tmp/sglang_weight_cache_rank6.sock"
      },
      {
        "ppRank": 0,
        "tpRank": 7,
        "globalRank": 7,
        "socket": "/tmp/sglang_weight_cache_rank7.sock"
      },
      {
        "ppRank": 1,
        "tpRank": 0,
        "globalRank": 8,
        "socket": "/tmp/sglang_weight_cache_rank8.sock"
      },
      {
        "ppRank": 1,
        "tpRank": 1,
        "globalRank": 9,
        "socket": "/tmp/sglang_weight_cache_rank9.sock"
      },
      {
        "ppRank": 1,
        "tpRank": 2,
        "globalRank": 10,
        "socket": "/tmp/sglang_weight_cache_rank10.sock"
      },
      {
        "ppRank": 1,
        "tpRank": 3,
        "globalRank": 11,
        "socket": "/tmp/sglang_weight_cache_rank11.sock"
      },
      {
        "ppRank": 1,
        "tpRank": 4,
        "globalRank": 12,
        "socket": "/tmp/sglang_weight_cache_rank12.sock"
      },
      {
        "ppRank": 1,
        "tpRank": 5,
        "globalRank": 13,
        "socket": "/tmp/sglang_weight_cache_rank13.sock"
      },
      {
        "ppRank": 1,
        "tpRank": 6,
        "globalRank": 14,
        "socket": "/tmp/sglang_weight_cache_rank14.sock"
      },
      {
        "ppRank": 1,
        "tpRank": 7,
        "globalRank": 15,
        "socket": "/tmp/sglang_weight_cache_rank15.sock"
      }
    ],
    "twoIndependentTp4Instances": {
      "instanceA": [
        {
          "ppRank": 0,
          "tpRank": 0,
          "globalRank": 0,
          "socket": "/tmp/sglang_weight_cache_rank0.sock"
        },
        {
          "ppRank": 0,
          "tpRank": 1,
          "globalRank": 1,
          "socket": "/tmp/sglang_weight_cache_rank1.sock"
        },
        {
          "ppRank": 0,
          "tpRank": 2,
          "globalRank": 2,
          "socket": "/tmp/sglang_weight_cache_rank2.sock"
        },
        {
          "ppRank": 0,
          "tpRank": 3,
          "globalRank": 3,
          "socket": "/tmp/sglang_weight_cache_rank3.sock"
        }
      ],
      "instanceB": [
        {
          "ppRank": 0,
          "tpRank": 0,
          "globalRank": 0,
          "socket": "/tmp/sglang_weight_cache_rank0.sock"
        },
        {
          "ppRank": 0,
          "tpRank": 1,
          "globalRank": 1,
          "socket": "/tmp/sglang_weight_cache_rank1.sock"
        },
        {
          "ppRank": 0,
          "tpRank": 2,
          "globalRank": 2,
          "socket": "/tmp/sglang_weight_cache_rank2.sock"
        },
        {
          "ppRank": 0,
          "tpRank": 3,
          "globalRank": 3,
          "socket": "/tmp/sglang_weight_cache_rank3.sock"
        }
      ],
      "collidingSockets": [
        "/tmp/sglang_weight_cache_rank0.sock",
        "/tmp/sglang_weight_cache_rank1.sock",
        "/tmp/sglang_weight_cache_rank2.sock",
        "/tmp/sglang_weight_cache_rank3.sock"
      ],
      "collisionCount": 4
    }
  },
  "quantizationMatrix": [
    {
      "artifact": "GLM-5.2 BF16",
      "quantMethod": "",
      "weightBlockSize": null,
      "sourceAllowlistResult": "code-eligible",
      "combinedGlm52WeightCacheEvidence": false
    },
    {
      "artifact": "zai-org/GLM-5.2-FP8",
      "quantMethod": "fp8",
      "weightBlockSize": [
        128,
        128
      ],
      "sourceAllowlistResult": "code-eligible",
      "combinedGlm52WeightCacheEvidence": false
    },
    {
      "artifact": "per-tensor FP8",
      "quantMethod": "fp8",
      "weightBlockSize": null,
      "sourceAllowlistResult": "hard-error",
      "combinedGlm52WeightCacheEvidence": false
    },
    {
      "artifact": "GLM-5.2 NVFP4",
      "quantMethod": "nvfp4",
      "weightBlockSize": null,
      "sourceAllowlistResult": "hard-error",
      "combinedGlm52WeightCacheEvidence": false
    },
    {
      "artifact": "GLM-5.2 MXFP4",
      "quantMethod": "mxfp4",
      "weightBlockSize": null,
      "sourceAllowlistResult": "hard-error",
      "combinedGlm52WeightCacheEvidence": false
    }
  ],
  "testCoverage": {
    "registeredWeightCacheModel": "Qwen/Qwen3-0.6B",
    "registeredWeightCacheTpSizes": [
      1,
      2
    ],
    "manualWeightCacheHarnessIsGeneric": true,
    "glm52Fp8Variants": [
      "TP8",
      "TP8+DP8",
      "TP8+DP8+MTP"
    ],
    "combinedGlm52WeightCacheTestsObserved": 0,
    "registeredWeightCacheVariantCount": 2,
    "glm52Fp8VariantCount": 3,
    "combinedCoverageGap": true
  },
  "publisherMeasurements": {
    "model": "Ling-2.6-1T FP8",
    "hardware": "8x H20-3e",
    "weightLoadBeforeSeconds": 495,
    "weightLoadAfterSeconds": 0.63,
    "reportedSpeedup": 785,
    "totalStartupBeforeMinutes": 8.8,
    "totalStartupAfterMinutes": 0.528,
    "transferableToGlm52": false,
    "computedWeightLoadRatio": 785.714,
    "computedTotalStartupReductionPercent": 94,
    "transferWarning": "Ling-2.6-1T publisher measurements are not GLM-5.2 measurements and must not be used as a GLM-5.2 latency forecast."
  },
  "boundaries": {
    "glm52WeightCacheRunPerformed": false,
    "glm52ParityEstablished": false,
    "glm52RestartLatencyMeasured": false,
    "multiInstanceNamespaceSafe": false,
    "dpEpShardKeyingComplete": false,
    "crossNodeLifecycleComplete": false,
    "mtpDraftWeightSupportComplete": false,
    "operationsMetricsAndSecurityComplete": false,
    "blogAndPinnedSourceFailureSemanticsAgree": false
  },
  "decision": {
    "classification": "narrow-lab-pilot-only",
    "eligibleCandidate": "Pinned SGLang v0.5.18, standalone daemon plus client, plain TP, no MTP or other speculative decoding, and either unquantized BF16 or the official 128x128 block-FP8 checkpoint.",
    "productionBlockers": [
      "No observed combined GLM-5.2 plus Weight Cache Daemon parity or restart-latency test.",
      "DP and EP shard identity fields are on the open roadmap rather than present in the pinned fingerprint.",
      "MTP and every speculative-decoding combination are rejected by pinned ServerArgs.",
      "Multi-instance namespace, cross-node lifecycle, metrics, and security hardening remain open roadmap work.",
      "Pinned client code hard-errors on config mismatch even though the publication describes a disk fallback."
    ]
  },
  "invariants": {
    "sourceReceiptCountIsNineteen": true,
    "fp8CheckpointIsBlockWise": true,
    "fp8CheckpointPassesSourceAllowlist": true,
    "mtpIsRejected": true,
    "missingDpEpNodeRanks": true,
    "tp8Pp1HasEightSockets": true,
    "tp8Pp2HasSixteenSockets": true,
    "twoTp4InstancesCollideOnFourSockets": true,
    "noCombinedGlmWeightCacheTest": true,
    "publisherNumbersAreNotTransferable": true,
    "blogAndCodeFailureSemanticsDiffer": true,
    "auditUsedNoRuntimeOrCredentials": true
  }
}
