{
  "schema_version": "1.0",
  "model": "GLM-5.2",
  "compiled_by": "GLM52.ai Editorial Team",
  "checked_at": "2026-07-20",
  "claim_boundary": "This is a structured snapshot of attributed third-party results, not a GLM52.ai model run.",
  "results": [
    {
      "evidence_type": "publisher_reported",
      "publisher": "Z.ai",
      "benchmark": "SWE-bench Pro",
      "metric": "score",
      "value": 62.1,
      "model_id": "GLM-5.2",
      "provider": "not stated in model card",
      "harness": "OpenHands",
      "context_tokens": 400000,
      "parameters": "temperature=1; top_p=1; max_new_tokens=32000",
      "source_date": "2026-06-16",
      "source_url": "https://huggingface.co/zai-org/GLM-5.2"
    },
    {
      "evidence_type": "publisher_reported",
      "publisher": "Z.ai",
      "benchmark": "DeepSWE",
      "metric": "score",
      "value": 46.2,
      "model_id": "GLM-5.2",
      "provider": "not stated in model card",
      "harness": "official pier framework + mini-swe-agent",
      "context_tokens": 400000,
      "parameters": "temperature=1; top_p=1; timeout=2h; 2 CPU; 8 GB RAM; no internet",
      "source_date": "2026-06-16",
      "source_url": "https://huggingface.co/zai-org/GLM-5.2"
    },
    {
      "evidence_type": "publisher_reported",
      "publisher": "Z.ai",
      "benchmark": "Terminal-Bench 2.1",
      "metric": "score",
      "value": 81.0,
      "model_id": "GLM-5.2",
      "provider": "not stated in model card",
      "harness": "Terminus-2",
      "context_tokens": 256000,
      "parameters": "parser=json; timeout=4h; temperature=1; top_p=1; max_new_tokens=48000; max_episodes=500; 4 CPU; 8 GB RAM",
      "source_date": "2026-06-16",
      "source_url": "https://huggingface.co/zai-org/GLM-5.2"
    },
    {
      "evidence_type": "publisher_reported",
      "publisher": "Z.ai",
      "benchmark": "MCP-Atlas public set",
      "metric": "score",
      "value": 76.8,
      "model_id": "GLM-5.2",
      "provider": "not stated in model card",
      "harness": "publisher evaluation",
      "context_tokens": null,
      "parameters": "think mode; 10-minute limit; Gemini-3-Pro judge",
      "source_date": "2026-06-16",
      "source_url": "https://huggingface.co/zai-org/GLM-5.2"
    },
    {
      "evidence_type": "independent_reported",
      "publisher": "Tessl",
      "benchmark": "Task evaluations for skills",
      "metric": "overall score with skill",
      "value": 91.9,
      "model_id": "GLM-5.2",
      "provider": "Fireworks Standard",
      "harness": "Tessl paired scenario evaluation",
      "context_tokens": null,
      "parameters": "same scenarios run in baseline and skill-assisted conditions",
      "source_date": "2026-06-18",
      "source_url": "https://tessl.io/blog/open-source-coding-agents-one-ties-sonnet-one-wont-listen/",
      "public_tasks_url": "https://huggingface.co/datasets/tesslio/task-evals-for-skills"
    },
    {
      "evidence_type": "independent_reported",
      "publisher": "Tessl",
      "benchmark": "Task evaluations for skills",
      "metric": "solve-only cost per task USD",
      "value": 0.289,
      "model_id": "GLM-5.2",
      "provider": "Fireworks Standard",
      "harness": "Tessl paired scenario evaluation",
      "context_tokens": null,
      "parameters": "measured token counts at published rates; grading excluded",
      "source_date": "2026-06-18",
      "source_url": "https://tessl.io/blog/open-source-coding-agents-one-ties-sonnet-one-wont-listen/"
    },
    {
      "evidence_type": "independent_reported",
      "publisher": "Entelligence",
      "benchmark": "45 selected Terminal-Bench tasks",
      "metric": "tasks passed",
      "value": 25,
      "denominator": 45,
      "model_id": "GLM-5.2",
      "provider": "not disclosed",
      "harness": "Claude Code",
      "context_tokens": null,
      "parameters": "40-turn budget; same agent, prompts, tools and hidden-test grading across compared models",
      "source_date": "2026-06-24",
      "source_url": "https://entelligence.ai/blogs/glm-5-2-vs-claude-opus-coding-benchmark"
    },
    {
      "evidence_type": "independent_reported",
      "publisher": "Artificial Analysis",
      "benchmark": "Artificial Analysis Intelligence Index v4.1",
      "metric": "index score",
      "value": 51,
      "model_id": "GLM-5.2 (max)",
      "provider": "current provider set",
      "harness": "Artificial Analysis evaluation suite",
      "context_tokens": null,
      "parameters": "current site methodology; consult source for revisions",
      "source_date": "2026-07-20",
      "source_url": "https://artificialanalysis.ai/models/glm-5-2"
    },
    {
      "evidence_type": "independent_reported",
      "publisher": "Artificial Analysis",
      "benchmark": "Provider performance median",
      "metric": "output tokens per second",
      "value": 192.3,
      "model_id": "GLM-5.2 (max)",
      "provider": "median across serving providers",
      "harness": "Artificial Analysis provider measurements",
      "context_tokens": null,
      "parameters": "current site methodology; not a Z.ai direct guarantee",
      "source_date": "2026-07-20",
      "source_url": "https://artificialanalysis.ai/models/glm-5-2"
    }
  ]
}
