{
  "checkedAt": "2026-08-19T21:16:47+08:00",
  "status": "publisher-benchmark-and-current-price-audit",
  "scope": {
    "authenticatedApiCalls": 0,
    "liveModelCalls": 0,
    "graderCalls": 0,
    "credentialsUsed": false,
    "description": "Source-contract audit and deterministic arithmetic only; no GLM-5.2 or GLM-5.3 inference was requested."
  },
  "benchmarkDeltas": [
    {
      "name": "Terminal-Bench 3.0",
      "unit": "score",
      "direction": "higher",
      "glm52": 4.6,
      "glm53": 28.3,
      "absoluteChange": 23.7,
      "relativeChangePercent": 515.2,
      "ratio": 6.152
    },
    {
      "name": "DeepSWE v1.1",
      "unit": "score",
      "direction": "higher",
      "glm52": 46.2,
      "glm53": 66.9,
      "absoluteChange": 20.7,
      "relativeChangePercent": 44.8,
      "ratio": 1.448
    },
    {
      "name": "Agents' Last Exam",
      "unit": "score",
      "direction": "higher",
      "glm52": 23.8,
      "glm53": 28.5,
      "absoluteChange": 4.7,
      "relativeChangePercent": 19.7,
      "ratio": 1.197
    },
    {
      "name": "Z.AI Code Bench Max",
      "unit": "percent",
      "direction": "higher",
      "glm52": 23.4,
      "glm53": 34.5,
      "absoluteChange": 11.1,
      "relativeChangePercent": 47.4,
      "ratio": 1.474
    },
    {
      "name": "Z.AI Code Bench Max output",
      "unit": "tokens",
      "direction": "lower",
      "glm52": 96000,
      "glm53": 75000,
      "absoluteChange": -21000,
      "relativeChangePercent": -21.9,
      "ratio": 0.781
    },
    {
      "name": "CyberGym",
      "unit": "percent",
      "direction": "higher",
      "glm52": 77.2,
      "glm53": 84.5,
      "absoluteChange": 7.3,
      "relativeChangePercent": 9.5,
      "ratio": 1.095
    },
    {
      "name": "ExploitBench",
      "unit": "percent",
      "direction": "higher",
      "glm52": 24.4,
      "glm53": 54.4,
      "absoluteChange": 30,
      "relativeChangePercent": 123,
      "ratio": 2.23
    },
    {
      "name": "ExploitGym two-hour budget",
      "unit": "tasks",
      "direction": "higher",
      "glm52": 29,
      "glm53": 105,
      "absoluteChange": 76,
      "relativeChangePercent": 262.1,
      "ratio": 3.621
    },
    {
      "name": "ExploitGym six-hour budget",
      "unit": "tasks",
      "direction": "higher",
      "glm52": 39,
      "glm53": 130,
      "absoluteChange": 91,
      "relativeChangePercent": 233.3,
      "ratio": 3.333
    }
  ],
  "scenarioCosts": [
    {
      "name": "bounded agent turn",
      "freshInputTokens": 100000,
      "cachedInputTokens": 20000,
      "outputTokens": 10000,
      "glm52": {
        "freshInputUsd": 0.14,
        "cachedInputUsd": 0.0052,
        "outputUsd": 0.044,
        "totalUsd": 0.1892
      },
      "glm53": {
        "freshInputUsd": 0.14,
        "cachedInputUsd": 0.0052,
        "outputUsd": 0.044,
        "totalUsd": 0.1892
      }
    },
    {
      "name": "repository-scale review",
      "freshInputTokens": 800000,
      "cachedInputTokens": 100000,
      "outputTokens": 50000,
      "glm52": {
        "freshInputUsd": 1.12,
        "cachedInputUsd": 0.026,
        "outputUsd": 0.22,
        "totalUsd": 1.366
      },
      "glm53": {
        "freshInputUsd": 1.12,
        "cachedInputUsd": 0.026,
        "outputUsd": 0.22,
        "totalUsd": 1.366
      }
    }
  ],
  "outputChargeIllustration": {
    "boundary": "Output-token charge only. Input, cache, retries, tools, and accepted-task rate are not supplied by this publisher row.",
    "glm52Usd": 0.4224,
    "glm53Usd": 0.33,
    "changeUsd": -0.0924,
    "changePercent": -21.9
  },
  "invariants": {
    "samePublishedUnitPrices": true,
    "samePublishedContext": true,
    "samePublishedMaximumOutput": true,
    "glm53ReasoningCannotBeDisabled": true,
    "allRowsFinite": true,
    "equalTokenScenariosHaveEqualCost": true,
    "zeroModelCalls": true
  },
  "limitations": [
    "All benchmark values are Z.AI publisher-reported and were not independently rerun for this audit.",
    "Rows are kept within their named harnesses; values from different benchmark versions are not subtracted.",
    "Lower output-token use on a private benchmark does not prove lower production cost per accepted task.",
    "The audit does not measure latency, throughput, tool-call accuracy, availability, or account-specific billing."
  ]
}
