{
  "schemaVersion": 1,
  "checkedAt": "2026-08-11T23:17:19+00:00",
  "method": {
    "containerImage": "python@sha256:0b29ab9e420820f53d1cd5ce0157dfe07bea8a7cff5b4754d6d95c07b0e5bc47",
    "networkMode": "host",
    "requests": 8,
    "metadataBytesRead": 20658716,
    "weightsDownloaded": false,
    "modelCalls": 0,
    "publishedPorts": 0,
    "cleanupVerified": true
  },
  "models": {
    "nemotronBF16": {
      "repo": "nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16",
      "revision": "63a200063804e06fdb41d6717e43bc92f67859d2",
      "license": "OpenMDW-1.1",
      "documentedRole": "customization reference",
      "artifact": {
        "indexedBytes": 65842365568,
        "decimalGB": 65.842,
        "binaryGiB": 61.32,
        "indexedSafetensorShards": 14
      },
      "config": {
        "maxPositionEmbeddings": 262144,
        "routedExperts": 128,
        "expertsPerToken": 6
      },
      "statuses": {
        "config": 200,
        "weightIndex": 200
      }
    },
    "nemotronNVFP4": {
      "repo": "nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4",
      "revision": "b14872a58afd94e2fbbc1e5a1fe4ebb7dfdc5fdd",
      "license": "OpenMDW-1.1",
      "documentedRole": "optimized inference",
      "artifact": {
        "indexedBytes": 21559589596,
        "decimalGB": 21.56,
        "binaryGiB": 20.079,
        "indexedSafetensorShards": 52
      },
      "config": {
        "maxPositionEmbeddings": 1048576,
        "routedExperts": 128,
        "expertsPerToken": 6
      },
      "statuses": {
        "config": 200,
        "weightIndex": 200
      }
    },
    "glm52BF16": {
      "repo": "zai-org/GLM-5.2",
      "revision": "b4734de4facf877f85769a911abafc5283eab3d9",
      "license": "MIT",
      "documentedRole": "full-precision checkpoint",
      "artifact": {
        "indexedBytes": 1506659919872,
        "decimalGB": 1506.66,
        "binaryGiB": 1403.186,
        "indexedSafetensorShards": 282
      },
      "config": {
        "maxPositionEmbeddings": 1048576,
        "routedExperts": 256,
        "expertsPerToken": 8
      },
      "statuses": {
        "config": 200,
        "weightIndex": 200
      }
    },
    "glm52FP8": {
      "repo": "zai-org/GLM-5.2-FP8",
      "revision": "ba978f7d347eaf65d22f1a86833408afdb953541",
      "license": "MIT",
      "documentedRole": "official reduced-precision checkpoint",
      "artifact": {
        "indexedBytes": 755617140416,
        "decimalGB": 755.617,
        "binaryGiB": 703.723,
        "indexedSafetensorShards": 141
      },
      "config": {
        "maxPositionEmbeddings": 1048576,
        "routedExperts": 256,
        "expertsPerToken": 8
      },
      "statuses": {
        "config": 200,
        "weightIndex": 200
      }
    }
  },
  "derived": {
    "glmToNemotronBF16IndexedByteRatio": 22.883,
    "glmFP8ToNemotronNVFP4IndexedByteRatio": 35.048,
    "nemotronBF16ToNVFP4IndexedByteRatio": 3.054,
    "glmBF16ToFP8IndexedByteRatio": 1.994,
    "nemotronNVFP4HeadroomInside32GiB": 11.921,
    "glmFP8Minimum80GiBDeviceCountByWeightsOnly": 9
  },
  "limits": [
    "Metadata-only audit; no model weights or inference were run.",
    "Indexed bytes are not peak VRAM, RAM, latency, throughput, quality, or usable context.",
    "The 32 GiB subtraction is capacity arithmetic, not proof that the runtime fits.",
    "Publisher benchmark tables were not normalized or rerun.",
    "License labels were checked against the model cards; this is not legal advice."
  ]
}
