{"apiVersion":"v1","evaluationSchemaVersion":"1.0.0","kind":"owner_reported_claim","record":{"id":"glm-5-2-gpqa-diamond-owner-reported","claimType":"owner_reported","modelId":"glm-5-2","modelName":"GLM-5.2","owner":"Z.ai","artifact":{"repository":"zai-org/GLM-5.2","revision":"b4734de4facf877f85769a911abafc5283eab3d9"},"artifactAssociation":"artifact_snapshot_associated","executionArtifactDigest":null,"benchmark":{"suite":"GPQA","version":null,"subset":"Diamond"},"metric":{"name":"owner-reported score","value":91.2,"unit":"reported_score","scale":"0-100 as presented","direction":"higher_is_better"},"evaluation":{"mode":"unspecified","reportedSettings":[{"name":"temperature","value":1},{"name":"top_p","value":0.95},{"name":"max_new_tokens","value":163840}]},"sourceRefs":[{"sourceId":"glm-5-2-pinned-model-card","locator":"Benchmark table, GPQA-Diamond row; Footnote for HLE and other reasoning tasks"}],"comparisonEligible":false,"missingContext":["The owner does not explicitly label the score as a percentage or define the metric scale.","Reasoning effort, prompt, shot count, sample count, and judge configuration are not reported for GPQA.","Benchmark revision, harness version, and hardware are not reported.","The provider does not publish a digest of the tensor artifact used for the evaluation."]},"sources":[{"id":"glm-5-2-pinned-model-card","title":"GLM-5.2 pinned model card","publisher":"Z.ai","url":"https://huggingface.co/zai-org/GLM-5.2/blob/b4734de4facf877f85769a911abafc5283eab3d9/README.md","retrievedAt":"2026-08-03","sourceType":"official_model_card","artifactSnapshot":{"repository":"zai-org/GLM-5.2","revision":"b4734de4facf877f85769a911abafc5283eab3d9"}}],"rawArtifacts":[]}