{"apiVersion":"v1","evaluationSchemaVersion":"1.0.0","kind":"owner_reported_claim","record":{"id":"kimi-k3-terminal-bench-2-1-owner-reported","claimType":"owner_reported","modelId":"kimi-k3","modelName":"Kimi K3","owner":"Moonshot AI","artifact":{"repository":"moonshotai/Kimi-K3","revision":"9f62e4e9fffbd0a83ddd60e1c209d828994b3569"},"artifactAssociation":"artifact_snapshot_associated","executionArtifactDigest":null,"benchmark":{"suite":"Terminal-Bench","version":"2.1","subset":null},"metric":{"name":"owner-reported score","value":88.3,"unit":"reported_score","scale":"0-100 as presented","direction":"higher_is_better"},"evaluation":{"mode":"max","reportedSettings":[{"name":"reasoning_effort","value":"max"},{"name":"temperature","value":1},{"name":"top_p","value":1},{"name":"harness","value":"Kimi Code"}]},"sourceRefs":[{"sourceId":"kimi-k3-pinned-model-card","locator":"Section 3, Evaluation Results, Terminal-Bench 2.1 row; Footnotes coding benchmarks"}],"comparisonEligible":false,"missingContext":["The owner does not explicitly label the score as a percentage or define the metric scale.","The Kimi Code harness version and full harness configuration are not reported.","Task revision, trial count, and aggregation method are not reported.","Inference engine and hardware are not reported.","The provider does not publish a digest of the tensor artifact used for the evaluation."]},"sources":[{"id":"kimi-k3-pinned-model-card","title":"Kimi K3 pinned model card","publisher":"Moonshot AI","url":"https://huggingface.co/moonshotai/Kimi-K3/blob/9f62e4e9fffbd0a83ddd60e1c209d828994b3569/README.md","retrievedAt":"2026-08-03","sourceType":"official_model_card","artifactSnapshot":{"repository":"moonshotai/Kimi-K3","revision":"9f62e4e9fffbd0a83ddd60e1c209d828994b3569"}}],"rawArtifacts":[]}