{"apiVersion":"v1","evaluationSchemaVersion":"1.0.0","schemaUrl":"https://moemodels.ai/schemas/evaluations-v1.json","generatedAt":"2026-08-03","filters":{"model":"qwen3-30b-a3b","suite":null,"artifactAssociation":null},"counts":{"reportedClaims":3,"normalizedRuns":0,"comparisonEligibleRuns":0},"adapters":[{"id":"lm-eval-v0-4-12","name":"EleutherAI lm-evaluation-harness","kind":"lm_eval","packageName":"lm-eval","version":"0.4.12","repositoryUrl":"https://github.com/EleutherAI/lm-evaluation-harness","revision":"6d642546f4688648fced259eb3302efd36ece5af","revisionUrl":"https://github.com/EleutherAI/lm-evaluation-harness/tree/6d642546f4688648fced259eb3302efd36ece5af","status":"pinned_not_executed","sourceIds":["lm-eval-v0-4-12-source"]}],"sources":[{"id":"kimi-k3-pinned-model-card","title":"Kimi K3 pinned model card","publisher":"Moonshot AI","url":"https://huggingface.co/moonshotai/Kimi-K3/blob/9f62e4e9fffbd0a83ddd60e1c209d828994b3569/README.md","retrievedAt":"2026-08-03","sourceType":"official_model_card","artifactSnapshot":{"repository":"moonshotai/Kimi-K3","revision":"9f62e4e9fffbd0a83ddd60e1c209d828994b3569"}},{"id":"deepseek-v4-pro-pinned-model-card","title":"DeepSeek V4 Pro pinned model card","publisher":"DeepSeek","url":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/b5968e9190ef611bbf34a7229255be88a0e937c1/README.md","retrievedAt":"2026-08-03","sourceType":"official_model_card","artifactSnapshot":{"repository":"deepseek-ai/DeepSeek-V4-Pro","revision":"b5968e9190ef611bbf34a7229255be88a0e937c1"}},{"id":"deepseek-v4-technical-report","title":"DeepSeek-V4: Towards Highly Efficient Million-Token Context Intelligence","publisher":"DeepSeek-AI","url":"https://arxiv.org/pdf/2606.19348v1","retrievedAt":"2026-08-03","sourceType":"technical_report"},{"id":"glm-5-2-pinned-model-card","title":"GLM-5.2 pinned model card","publisher":"Z.ai","url":"https://huggingface.co/zai-org/GLM-5.2/blob/b4734de4facf877f85769a911abafc5283eab3d9/README.md","retrievedAt":"2026-08-03","sourceType":"official_model_card","artifactSnapshot":{"repository":"zai-org/GLM-5.2","revision":"b4734de4facf877f85769a911abafc5283eab3d9"}},{"id":"gemma-4-26b-a4b-it-pinned-model-card","title":"Gemma 4 26B A4B IT pinned model card","publisher":"Google DeepMind","url":"https://huggingface.co/google/gemma-4-26B-A4B-it/blob/4d7ae4984b7db7de8f8457170b3f1a419ee76d52/README.md","retrievedAt":"2026-08-03","sourceType":"official_model_card","artifactSnapshot":{"repository":"google/gemma-4-26B-A4B-it","revision":"4d7ae4984b7db7de8f8457170b3f1a419ee76d52"}},{"id":"gemma-4-technical-report","title":"Gemma 4 Technical Report","publisher":"Gemma Team, Google DeepMind","url":"https://arxiv.org/pdf/2607.02770v1","retrievedAt":"2026-08-03","sourceType":"technical_report"},{"id":"qwen3-30b-a3b-pinned-model-card","title":"Qwen3-30B-A3B pinned model card","publisher":"Qwen Team","url":"https://huggingface.co/Qwen/Qwen3-30B-A3B/blob/ad44e777bcd18fa416d9da3bd8f70d33ebb85d39/README.md","retrievedAt":"2026-08-03","sourceType":"official_model_card","artifactSnapshot":{"repository":"Qwen/Qwen3-30B-A3B","revision":"ad44e777bcd18fa416d9da3bd8f70d33ebb85d39"}},{"id":"qwen3-technical-report","title":"Qwen3 Technical Report","publisher":"Qwen Team","url":"https://arxiv.org/pdf/2505.09388v1","retrievedAt":"2026-08-03","sourceType":"technical_report"},{"id":"lm-eval-v0-4-12-source","title":"EleutherAI lm-evaluation-harness v0.4.12 pinned source","publisher":"EleutherAI","url":"https://github.com/EleutherAI/lm-evaluation-harness/tree/6d642546f4688648fced259eb3302efd36ece5af","retrievedAt":"2026-08-03","sourceType":"adapter_repository"}],"reportedClaims":[{"id":"qwen3-30b-a3b-gpqa-diamond-owner-reported","claimType":"owner_reported","modelId":"qwen3-30b-a3b","modelName":"Qwen3-30B-A3B","owner":"Qwen Team","artifact":{"repository":"Qwen/Qwen3-30B-A3B","revision":"ad44e777bcd18fa416d9da3bd8f70d33ebb85d39"},"artifactAssociation":"model_name_only","executionArtifactDigest":null,"benchmark":{"suite":"GPQA","version":null,"subset":"Diamond"},"metric":{"name":"averaged accuracy","value":65.8,"unit":"percentage_points","scale":"0-100","direction":"higher_is_better"},"evaluation":{"mode":"thinking","reportedSettings":[{"name":"thinking_mode","value":true},{"name":"temperature","value":0.6},{"name":"top_p","value":0.95},{"name":"top_k","value":20},{"name":"max_new_tokens","value":32768},{"name":"samples_per_query","value":10}]},"sourceRefs":[{"sourceId":"qwen3-technical-report","locator":"Section 4.2 Evaluation Setup and Table 15, Qwen3-30B-A3B Thinking column"},{"sourceId":"qwen3-30b-a3b-pinned-model-card","locator":"Pinned artifact identity; the card links to general benchmark material but contains no values"}],"comparisonEligible":false,"missingContext":["The technical report identifies the model by name but does not bind the result to artifact revision ad44e777bcd18fa416d9da3bd8f70d33ebb85d39.","Exact prompt, benchmark revision, and evaluation harness are not reported.","Inference engine and hardware are not reported.","The provider does not publish a digest of the tensor artifact used for the evaluation."]},{"id":"qwen3-30b-a3b-aime-2025-owner-reported","claimType":"owner_reported","modelId":"qwen3-30b-a3b","modelName":"Qwen3-30B-A3B","owner":"Qwen Team","artifact":{"repository":"Qwen/Qwen3-30B-A3B","revision":"ad44e777bcd18fa416d9da3bd8f70d33ebb85d39"},"artifactAssociation":"model_name_only","executionArtifactDigest":null,"benchmark":{"suite":"AIME","version":"2025","subset":"Parts I and II"},"metric":{"name":"averaged accuracy","value":70.9,"unit":"percentage_points","scale":"0-100","direction":"higher_is_better"},"evaluation":{"mode":"thinking","reportedSettings":[{"name":"thinking_mode","value":true},{"name":"temperature","value":0.6},{"name":"top_p","value":0.95},{"name":"top_k","value":20},{"name":"max_new_tokens","value":38912},{"name":"question_count","value":30},{"name":"samples_per_query","value":64}]},"sourceRefs":[{"sourceId":"qwen3-technical-report","locator":"Section 4.2 Evaluation Setup and Table 15, Qwen3-30B-A3B Thinking column, AIME 2025 row"},{"sourceId":"qwen3-30b-a3b-pinned-model-card","locator":"Pinned artifact identity; the card links to general benchmark material but contains no values"}],"comparisonEligible":false,"missingContext":["The technical report identifies the model by name but does not bind the result to artifact revision ad44e777bcd18fa416d9da3bd8f70d33ebb85d39.","Exact prompt text, scoring parser, and test-data revision are not reported.","Inference engine and hardware are not reported.","The provider does not publish a digest of the tensor artifact used for the evaluation."]},{"id":"qwen3-30b-a3b-livecodebench-v5-owner-reported","claimType":"owner_reported","modelId":"qwen3-30b-a3b","modelName":"Qwen3-30B-A3B","owner":"Qwen Team","artifact":{"repository":"Qwen/Qwen3-30B-A3B","revision":"ad44e777bcd18fa416d9da3bd8f70d33ebb85d39"},"artifactAssociation":"model_name_only","executionArtifactDigest":null,"benchmark":{"suite":"LiveCodeBench","version":"v5","subset":"2024-10 through 2025-02"},"metric":{"name":"owner-reported score","value":62.6,"unit":"reported_score","scale":"0-100 as presented","direction":"higher_is_better"},"evaluation":{"mode":"thinking","reportedSettings":[{"name":"thinking_mode","value":true},{"name":"temperature","value":0.6},{"name":"top_p","value":0.95},{"name":"top_k","value":20},{"name":"max_new_tokens","value":32768},{"name":"benchmark_window","value":"2024-10 through 2025-02"},{"name":"prompt_variant","value":"official prompt with the program-only restriction removed"}]},"sourceRefs":[{"sourceId":"qwen3-technical-report","locator":"Section 4.2 Evaluation Setup and Table 15, Qwen3-30B-A3B Thinking column, LiveCodeBench v5 row"},{"sourceId":"qwen3-30b-a3b-pinned-model-card","locator":"Pinned artifact identity; the card links to general benchmark material but contains no values"}],"comparisonEligible":false,"missingContext":["The technical report identifies the model by name but does not bind the result to artifact revision ad44e777bcd18fa416d9da3bd8f70d33ebb85d39.","The report does not explicitly define the table's LiveCodeBench metric or sample count.","Exact task commit, harness version, inference engine, and hardware are not reported.","The provider does not publish a digest of the tensor artifact used for the evaluation."]}],"rawArtifacts":[],"runs":[]}